Micron Document
<!DOCTYPE html>
<html class="client-nojs vector-feature-night-mode-disabled vector-feature-language-in-header-enabled vector-feature-language-in-main-page-header-disabled vector-feature-page-tools-pinned-disabled vector-feature-toc-pinned-clientpref-1 vector-feature-main-menu-pinned-disabled vector-feature-limited-width-clientpref-1 vector-feature-limited-width-content-enabled vector-feature-custom-font-size-clientpref-1 vector-feature-appearance-pinned-clientpref-1 vector-sticky-header-enabled" lang="en" dir="ltr"><head>
<meta charset="UTF-8">
<title>Large language model</title>
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="canonical" href="https://en.wikipedia.org/wiki/Large_language_model"> <link href="./mw/ext.cite.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/ext.math.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.icons.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.search.codex.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/user.styles.css" rel="stylesheet" type="text/css">
<meta name="ResourceLoaderDynamicStyles" content="">
<link rel="stylesheet" type="text/css" href="./mw/site.styles.css">
<link rel="stylesheet" type="text/css" href="./mw/noscript.css">
<link rel="stylesheet" type="text/css" href="./footer.css">
<link rel="stylesheet" type="text/css" href="./vector-2022.css">
</head>
<body class="skin--responsive skin-vector skin-vector-search-vue mediawiki ltr sitedir-ltr mw-hide-empty-elt ns-0 ns-subject page-Large_language_model rootpage-Large_language_model skin-vector-2022 action-view">
<div class="mw-page-container">
<div class="mw-page-container-inner">
<div class="mw-content-container">
<main id="content" class="mw-body">
<header class="mw-body-header vector-page-titlebar">
<h1 id="firstHeading" class="firstHeading mw-first-heading">
<span id="openzim-page-title" class="mw-page-title-main"><span class="mw-page-title-main">Large language model</span></span>
</h1>
</header>
<a id="top"></a>
<div id="bodyContent" class="vector-body ve-init-mw-desktopArticleTarget-targetContainer" aria-labelledby="firstHeading" data-mw-ve-target-container="">
<div id="mw-content-text" class="mw-body-content mw-content-ltr" lang="en" dir="ltr"><div class="mw-content-ltr mw-parser-output" lang="en" dir="ltr">
<style data-mw-deduplicate="TemplateStyles:r1236090951">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hatnote{font-style:italic}.mw-parser-output div.hatnote{padding-left:1.6em;margin-bottom:0.5em}.mw-parser-output .hatnote i{font-style:normal}.mw-parser-output .hatnote+link+.hatnote{margin-top:-0.5em}@media print{body.ns-0 .mw-parser-output .hatnote{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><div role="note" class="hatnote navigation-not-searchable">Not to be confused with <a href="Logic_learning_machine" title="Logic learning machine">Logic learning machine</a>.</div>
<div role="note" class="hatnote navigation-not-searchable">"LLM" redirects here. For other uses, see <a href="LLM_(disambiguation)" class="mw-disambig" title="LLM (disambiguation)">LLM (disambiguation)</a>.</div>
<style data-mw-deduplicate="TemplateStyles:r1305433154">
/* start https://en.wikipedia.org/ */


.mw-parser-output .ambox{border:1px solid #a2a9b1;border-left:10px solid #36c;background-color:#fbfbfb;box-sizing:border-box}.mw-parser-output .ambox+link+.ambox,.mw-parser-output .ambox+link+style+.ambox,.mw-parser-output .ambox+link+link+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+style+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+link+.ambox{margin-top:-1px}html body.mediawiki .mw-parser-output .ambox.mbox-small-left{margin:4px 1em 4px 0;overflow:hidden;width:238px;border-collapse:collapse;font-size:88%;line-height:1.25em}.mw-parser-output .ambox-speedy{border-left:10px solid #b32424;background-color:#fee7e6}.mw-parser-output .ambox-delete{border-left:10px solid #b32424}.mw-parser-output .ambox-content{border-left:10px solid #f28500}.mw-parser-output .ambox-style{border-left:10px solid #fc3}.mw-parser-output .ambox-move{border-left:10px solid #9932cc}.mw-parser-output .ambox-protection{border-left:10px solid #a2a9b1}.mw-parser-output .ambox .mbox-text{border:none;padding:0.25em 0.5em;width:100%}.mw-parser-output .ambox .mbox-image{border:none;padding:2px 0 2px 0.5em;text-align:center}.mw-parser-output .ambox .mbox-imageright{border:none;padding:2px 0.5em 2px 0;text-align:center}.mw-parser-output .ambox .mbox-empty-cell{border:none;padding:0;width:1px}.mw-parser-output .ambox .mbox-image-div{width:52px}@media(min-width:720px){.mw-parser-output .ambox{margin:0 10%}}@media print{body.ns-0 .mw-parser-output .ambox{display:none!important}}


/* end https://en.wikipedia.org/ */
</style>
<style data-mw-deduplicate="TemplateStyles:r1129693374">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hlist dl,.mw-parser-output .hlist ol,.mw-parser-output .hlist ul{margin:0;padding:0}.mw-parser-output .hlist dd,.mw-parser-output .hlist dt,.mw-parser-output .hlist li{margin:0;display:inline}.mw-parser-output .hlist.inline,.mw-parser-output .hlist.inline dl,.mw-parser-output .hlist.inline ol,.mw-parser-output .hlist.inline ul,.mw-parser-output .hlist dl dl,.mw-parser-output .hlist dl ol,.mw-parser-output .hlist dl ul,.mw-parser-output .hlist ol dl,.mw-parser-output .hlist ol ol,.mw-parser-output .hlist ol ul,.mw-parser-output .hlist ul dl,.mw-parser-output .hlist ul ol,.mw-parser-output .hlist ul ul{display:inline}.mw-parser-output .hlist .mw-empty-li{display:none}.mw-parser-output .hlist dt::after{content:": "}.mw-parser-output .hlist dd::after,.mw-parser-output .hlist li::after{content:" · ";font-weight:bold}.mw-parser-output .hlist dd:last-child::after,.mw-parser-output .hlist dt:last-child::after,.mw-parser-output .hlist li:last-child::after{content:none}.mw-parser-output .hlist dd dd:first-child::before,.mw-parser-output .hlist dd dt:first-child::before,.mw-parser-output .hlist dd li:first-child::before,.mw-parser-output .hlist dt dd:first-child::before,.mw-parser-output .hlist dt dt:first-child::before,.mw-parser-output .hlist dt li:first-child::before,.mw-parser-output .hlist li dd:first-child::before,.mw-parser-output .hlist li dt:first-child::before,.mw-parser-output .hlist li li:first-child::before{content:" (";font-weight:normal}.mw-parser-output .hlist dd dd:last-child::after,.mw-parser-output .hlist dd dt:last-child::after,.mw-parser-output .hlist dd li:last-child::after,.mw-parser-output .hlist dt dd:last-child::after,.mw-parser-output .hlist dt dt:last-child::after,.mw-parser-output .hlist dt li:last-child::after,.mw-parser-output .hlist li dd:last-child::after,.mw-parser-output .hlist li dt:last-child::after,.mw-parser-output .hlist li li:last-child::after{content:")";font-weight:normal}.mw-parser-output .hlist ol{counter-reset:listitem}.mw-parser-output .hlist ol>li{counter-increment:listitem}.mw-parser-output .hlist ol>li::before{content:" "counter(listitem)"\a0 "}.mw-parser-output .hlist dd ol>li:first-child::before,.mw-parser-output .hlist dt ol>li:first-child::before,.mw-parser-output .hlist li ol>li:first-child::before{content:" ("counter(listitem)"\a0 "}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r1246091330">
/* start https://en.wikipedia.org/ */


.mw-parser-output .sidebar{width:22em;float:right;clear:right;margin:0.5em 0 1em 1em;background:var(--background-color-neutral-subtle,#f8f9fa);border:1px solid var(--border-color-base,#a2a9b1);padding:0.2em;text-align:center;line-height:1.4em;font-size:88%;border-collapse:collapse;display:table}body.skin-minerva .mw-parser-output .sidebar{display:table!important;float:right!important;margin:0.5em 0 1em 1em!important}.mw-parser-output .sidebar-subgroup{width:100%;margin:0;border-spacing:0}.mw-parser-output .sidebar-left{float:left;clear:left;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-none{float:none;clear:both;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-outer-title{padding:0 0.4em 0.2em;font-size:125%;line-height:1.2em;font-weight:bold}.mw-parser-output .sidebar-top-image{padding:0.4em}.mw-parser-output .sidebar-top-caption,.mw-parser-output .sidebar-pretitle-with-top-image,.mw-parser-output .sidebar-caption{padding:0.2em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-pretitle{padding:0.4em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-title,.mw-parser-output .sidebar-title-with-pretitle{padding:0.2em 0.8em;font-size:145%;line-height:1.2em}.mw-parser-output .sidebar-title-with-pretitle{padding:0.1em 0.4em}.mw-parser-output .sidebar-image{padding:0.2em 0.4em 0.4em}.mw-parser-output .sidebar-heading{padding:0.1em 0.4em}.mw-parser-output .sidebar-content{padding:0 0.5em 0.4em}.mw-parser-output .sidebar-content-with-subgroup{padding:0.1em 0.4em 0.2em}.mw-parser-output .sidebar-above,.mw-parser-output .sidebar-below{padding:0.3em 0.8em;font-weight:bold}.mw-parser-output .sidebar-collapse .sidebar-above,.mw-parser-output .sidebar-collapse .sidebar-below{border-top:1px solid #aaa;border-bottom:1px solid #aaa}.mw-parser-output .sidebar-navbar{text-align:right;font-size:115%;padding:0 0.4em 0.4em}.mw-parser-output .sidebar-list-title{padding:0 0.4em;text-align:left;font-weight:bold;line-height:1.6em;font-size:105%}.mw-parser-output .sidebar-list-title-c{padding:0 0.4em;text-align:center;margin:0 3.3em}@media(max-width:640px){body.mediawiki .mw-parser-output .sidebar{width:100%!important;clear:both;float:none!important;margin-left:0!important;margin-right:0!important}}body.skin--responsive .mw-parser-output .sidebar a>img{max-width:none!important}@media screen{html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media print{body.ns-0 .mw-parser-output .sidebar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r886047488">
/* start https://en.wikipedia.org/ */


.mw-parser-output .nobold{font-weight:normal}


/* end https://en.wikipedia.org/ */
</style><table class="sidebar sidebar-collapse nomobile nowraplinks"><tbody><tr><td class="sidebar-pretitle">Part of a series on</td></tr><tr><th class="sidebar-title-with-pretitle"><a href="Machine_learning" title="Machine learning">Machine learning</a><br>and <a href="Data_mining" title="Data mining">data mining</a></th></tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Paradigms</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Supervised_learning" title="Supervised learning">Supervised learning</a></li>
<li><a href="Unsupervised_learning" title="Unsupervised learning">Unsupervised learning</a></li>
<li><a href="Semi-supervised_learning" class="mw-redirect" title="Semi-supervised learning">Semi-supervised learning</a></li>
<li><a href="Self-supervised_learning" title="Self-supervised learning">Self-supervised learning</a></li>
<li><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></li>
<li><a href="Meta-learning_(computer_science)" title="Meta-learning (computer science)">Meta-learning</a></li>
<li><a href="Online_machine_learning" title="Online machine learning">Online learning</a></li>
<li><a href="Batch_learning" class="mw-redirect" title="Batch learning">Batch learning</a></li>
<li><a href="Curriculum_learning" title="Curriculum learning">Curriculum learning</a></li>
<li><a href="Rule-based_machine_learning" title="Rule-based machine learning">Rule-based learning</a></li>
<li><a href="Neuro-symbolic_AI" title="Neuro-symbolic AI">Neuro-symbolic AI</a></li>
<li><a href="Neuromorphic_engineering" class="mw-redirect" title="Neuromorphic engineering">Neuromorphic engineering</a></li>
<li><a href="Quantum_machine_learning" title="Quantum machine learning">Quantum machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Problems</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Statistical_classification" title="Statistical classification">Classification</a></li>
<li><a href="Generative_model" title="Generative model">Generative modeling</a></li>
<li><a href="Regression_analysis" title="Regression analysis">Regression</a></li>
<li><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></li>
<li><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></li>
<li><a href="Density_estimation" title="Density estimation">Density estimation</a></li>
<li><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></li>
<li><a href="Data_cleaning" class="mw-redirect" title="Data cleaning">Data cleaning</a></li>
<li><a href="Automated_machine_learning" title="Automated machine learning">AutoML</a></li>
<li><a href="Association_rule_learning" title="Association rule learning">Association rules</a></li>
<li><a href="Semantic_analysis_(machine_learning)" title="Semantic analysis (machine learning)">Semantic analysis</a></li>
<li><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></li>
<li><a href="Feature_engineering" title="Feature engineering">Feature engineering</a></li>
<li><a href="Feature_learning" title="Feature learning">Feature learning</a></li>
<li><a href="Learning_to_rank" title="Learning to rank">Learning to rank</a></li>
<li><a href="Grammar_induction" title="Grammar induction">Grammar induction</a></li>
<li><a href="Ontology_learning" title="Ontology learning">Ontology learning</a></li>
<li><a href="Multimodal_learning" title="Multimodal learning">Multimodal learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><div style="display: inline-block; line-height: 1.2em; padding: .1em 0;"><a href="Supervised_learning" title="Supervised learning">Supervised learning</a><br><span class="nobold"><span style="font-size: 85%;">(<b><a href="Statistical_classification" title="Statistical classification">classification</a></b>&nbsp;• <b><a href="Regression_analysis" title="Regression analysis">regression</a></b>)</span></span> </div></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Apprenticeship_learning" title="Apprenticeship learning">Apprenticeship learning</a></li>
<li><a href="Decision_tree_learning" title="Decision tree learning">Decision trees</a></li>
<li><a href="Ensemble_learning" title="Ensemble learning">Ensembles</a>
<ul><li><a href="Bootstrap_aggregating" title="Bootstrap aggregating">Bagging</a></li>
<li><a href="Boosting_(machine_learning)" title="Boosting (machine learning)">Boosting</a></li>
<li><a href="Random_forest" title="Random forest">Random forest</a></li></ul></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Linear_regression" title="Linear regression">Linear regression</a></li>
<li><a href="Naive_Bayes_classifier" title="Naive Bayes classifier">Naive Bayes</a></li>
<li><a href="Artificial_neural_network" class="mw-redirect" title="Artificial neural network">Artificial neural networks</a></li>
<li><a href="Logistic_regression" title="Logistic regression">Logistic regression</a></li>
<li><a href="Perceptron" title="Perceptron">Perceptron</a></li>
<li><a href="Relevance_vector_machine" title="Relevance vector machine">Relevance vector machine (RVM)</a></li>
<li><a href="Support_vector_machine" title="Support vector machine">Support vector machine (SVM)</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="BIRCH" title="BIRCH">BIRCH</a></li>
<li><a href="CURE_algorithm" title="CURE algorithm">CURE</a></li>
<li><a href="Hierarchical_clustering" title="Hierarchical clustering">Hierarchical</a></li>
<li><a href="K-means_clustering" title="K-means clustering"><i>k</i>-means</a></li>
<li><a href="Fuzzy_clustering" title="Fuzzy clustering">Fuzzy</a></li>
<li><a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization (EM)</a></li>
<li><br><a href="DBSCAN" title="DBSCAN">DBSCAN</a></li>
<li><a href="OPTICS_algorithm" title="OPTICS algorithm">OPTICS</a></li>
<li><a href="Mean_shift" title="Mean shift">Mean shift</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Factor_analysis" title="Factor analysis">Factor analysis</a></li>
<li><a href="Canonical_correlation" title="Canonical correlation">CCA</a></li>
<li><a href="Independent_component_analysis" title="Independent component analysis">ICA</a></li>
<li><a href="Linear_discriminant_analysis" title="Linear discriminant analysis">LDA</a></li>
<li><a href="Non-negative_matrix_factorization" title="Non-negative matrix factorization">NMF</a></li>
<li><a href="Principal_component_analysis" title="Principal component analysis">PCA</a></li>
<li><a href="Proper_generalized_decomposition" title="Proper generalized decomposition">PGD</a></li>
<li><a href="T-distributed_stochastic_neighbor_embedding" title="T-distributed stochastic neighbor embedding">t-SNE</a></li>
<li><a href="Sparse_dictionary_learning" title="Sparse dictionary learning">SDL</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Graphical_model" title="Graphical model">Graphical models</a>
<ul><li><a href="Bayesian_network" title="Bayesian network">Bayes net</a></li>
<li><a href="Conditional_random_field" title="Conditional random field">Conditional random field</a></li>
<li><a href="Hidden_Markov_model" title="Hidden Markov model">Hidden Markov</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Random_sample_consensus" title="Random sample consensus">RANSAC</a></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Local_outlier_factor" title="Local outlier factor">Local outlier factor</a></li>
<li><a href="Isolation_forest" title="Isolation forest">Isolation forest</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Neural_network_(machine_learning)" title="Neural network (machine learning)">Neural networks</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Autoencoder" title="Autoencoder">Autoencoder</a></li>
<li><a href="Deep_learning" title="Deep learning">Deep learning</a></li>
<li><a href="Feedforward_neural_network" title="Feedforward neural network">Feedforward neural network</a></li>
<li><a href="Recurrent_neural_network" title="Recurrent neural network">Recurrent neural network</a>
<ul><li><a href="Long_short-term_memory" title="Long short-term memory">LSTM</a></li>
<li><a href="Gated_recurrent_unit" title="Gated recurrent unit">GRU</a></li>
<li><a href="Echo_state_network" title="Echo state network">ESN</a></li>
<li><a href="Reservoir_computing" title="Reservoir computing">reservoir computing</a></li></ul></li>
<li><a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann machine</a>
<ul><li><a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">Restricted</a></li></ul></li>
<li><a href="Generative_adversarial_network" title="Generative adversarial network">GAN</a></li>
<li><a href="Diffusion_model" title="Diffusion model">Diffusion model</a></li>
<li><a href="Self-organizing_map" title="Self-organizing map">SOM</a></li>
<li><a href="Convolutional_neural_network" title="Convolutional neural network">Convolutional neural network</a>
<ul><li><a href="U-Net" title="U-Net">U-Net</a></li>
<li><a href="LeNet" title="LeNet">LeNet</a></li>
<li><a href="AlexNet" title="AlexNet">AlexNet</a></li>
<li><a href="DeepDream" title="DeepDream">DeepDream</a></li></ul></li>
<li><a href="Neural_field" title="Neural field">Neural field</a>
<ul><li><a href="Neural_radiance_field" title="Neural radiance field">Neural radiance field</a></li>
<li><a href="Physics-informed_neural_networks" title="Physics-informed neural networks">Physics-informed neural networks</a></li></ul></li>
<li><a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">Transformer</a>
<ul><li><a href="Vision_transformer" title="Vision transformer">Vision</a></li></ul></li>
<li><a href="Mamba_(deep_learning_architecture)" title="Mamba (deep learning architecture)">Mamba</a></li>
<li><a href="Spiking_neural_network" title="Spiking neural network">Spiking neural network</a></li>
<li><a href="Memtransistor" title="Memtransistor">Memtransistor</a></li>
<li><a href="Electrochemical_RAM" title="Electrochemical RAM">Electrochemical RAM</a> (ECRAM)</li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Q-learning" title="Q-learning">Q-learning</a></li>
<li><a href="Policy_gradient_method" title="Policy gradient method">Policy gradient</a></li>
<li><a href="State%E2%80%93action%E2%80%93reward%E2%80%93state%E2%80%93action" title="State–action–reward–state–action">SARSA</a></li>
<li><a href="Temporal_difference_learning" title="Temporal difference learning">Temporal difference (TD)</a></li>
<li><a href="Multi-agent_reinforcement_learning" title="Multi-agent reinforcement learning">Multi-agent</a>
<ul><li><a href="Self-play_(reinforcement_learning_technique)" class="mw-redirect" title="Self-play (reinforcement learning technique)">Self-play</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Learning with humans</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Active_learning_(machine_learning)" title="Active learning (machine learning)">Active learning</a></li>
<li><a href="Crowdsourcing" title="Crowdsourcing">Crowdsourcing</a></li>
<li><a href="Human-in-the-loop" title="Human-in-the-loop">Human-in-the-loop</a></li>
<li><a href="Mechanistic_interpretability" title="Mechanistic interpretability">Mechanistic interpretability</a></li>
<li><a href="Reinforcement_learning_from_human_feedback" title="Reinforcement learning from human feedback">RLHF</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Model diagnostics</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Coefficient_of_determination" title="Coefficient of determination">Coefficient of determination</a></li>
<li><a href="Confusion_matrix" title="Confusion matrix">Confusion matrix</a></li>
<li><a href="Learning_curve_(machine_learning)" title="Learning curve (machine learning)">Learning curve</a></li>
<li><a href="Receiver_operating_characteristic" title="Receiver operating characteristic">ROC curve</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Mathematical foundations</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Kernel_machines" class="mw-redirect" title="Kernel machines">Kernel machines</a></li>
<li><a href="Bias%E2%80%93variance_tradeoff" title="Bias–variance tradeoff">Bias–variance tradeoff</a></li>
<li><a href="Computational_learning_theory" title="Computational learning theory">Computational learning theory</a></li>
<li><a href="Empirical_risk_minimization" title="Empirical risk minimization">Empirical risk minimization</a></li>
<li><a href="Occam_learning" title="Occam learning">Occam learning</a></li>
<li><a href="Probably_approximately_correct_learning" title="Probably approximately correct learning">PAC learning</a></li>
<li><a href="Statistical_learning_theory" title="Statistical learning theory">Statistical learning</a></li>
<li><a href="Vapnik%E2%80%93Chervonenkis_theory" title="Vapnik–Chervonenkis theory">VC theory</a></li>
<li><a href="Topological_deep_learning" title="Topological deep learning">Topological deep learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Journals and conferences</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="AAAI_Conference_on_Artificial_Intelligence" title="AAAI Conference on Artificial Intelligence">AAAI</a></li>
<li><a href="ECML_PKDD" title="ECML PKDD">ECML PKDD</a></li>
<li><a href="Conference_on_Neural_Information_Processing_Systems" title="Conference on Neural Information Processing Systems">NeurIPS</a></li>
<li><a href="International_Conference_on_Machine_Learning" title="International Conference on Machine Learning">ICML</a></li>
<li><a href="International_Conference_on_Learning_Representations" title="International Conference on Learning Representations">ICLR</a></li>
<li><a href="International_Joint_Conference_on_Artificial_Intelligence" title="International Joint Conference on Artificial Intelligence">IJCAI</a></li>
<li><a href="Machine_Learning_(journal)" title="Machine Learning (journal)">ML</a></li>
<li><a href="Journal_of_Machine_Learning_Research" title="Journal of Machine Learning Research">JMLR</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Related articles</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Glossary_of_artificial_intelligence" title="Glossary of artificial intelligence">Glossary of artificial intelligence</a></li>
<li><a href="List_of_datasets_for_machine-learning_research" title="List of datasets for machine-learning research">List of datasets for machine-learning research</a>
<ul><li><a href="List_of_datasets_in_computer_vision_and_image_processing" title="List of datasets in computer vision and image processing">List of datasets in computer vision and image processing</a></li></ul></li>
<li><a href="Outline_of_machine_learning" title="Outline of machine learning">Outline of machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-navbar"><style data-mw-deduplicate="TemplateStyles:r1239400231">
/* start https://en.wikipedia.org/ */


.mw-parser-output .navbar{display:inline;font-size:88%;font-weight:normal}.mw-parser-output .navbar-collapse{float:left;text-align:left}.mw-parser-output .navbar-boxtext{word-spacing:0}.mw-parser-output .navbar ul{display:inline-block;white-space:nowrap;line-height:inherit}.mw-parser-output .navbar-brackets::before{margin-right:-0.125em;content:"[ "}.mw-parser-output .navbar-brackets::after{margin-left:-0.125em;content:" ]"}.mw-parser-output .navbar li{word-spacing:-0.125em}.mw-parser-output .navbar a>span,.mw-parser-output .navbar a>abbr{text-decoration:inherit}.mw-parser-output .navbar-mini abbr{font-variant:small-caps;border-bottom:none;text-decoration:none;cursor:inherit}.mw-parser-output .navbar-ct-full{font-size:114%;margin:0 7em}.mw-parser-output .navbar-ct-mini{font-size:114%;margin:0 4em}html.skin-theme-clientpref-night .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}@media(prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}}@media print{.mw-parser-output .navbar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style></td></tr></tbody></table><p>A <b>large language model</b> (<b>LLM</b>) is a <a href="Language_model" title="Language model">language model</a> trained with <a href="Self-supervised_learning" title="Self-supervised learning">self-supervised</a> <a href="Machine_learning" title="Machine learning">machine learning</a> on a vast amount of text, designed for <a href="Natural_language_processing" title="Natural language processing">natural language processing</a> tasks, especially <a href="Natural_language_generation" title="Natural language generation">language generation</a>.
</p><p>The largest and most capable LLMs are <a href="Generative_pre-trained_transformer" title="Generative pre-trained transformer">generative pretrained transformers</a> (GPTs), which are largely used in <a href="Generative_artificial_intelligence" title="Generative artificial intelligence">generative</a> <a href="Chatbot" title="Chatbot">chatbots</a> such as <a href="ChatGPT" title="ChatGPT">ChatGPT</a>, <a href="Gemini_(chatbot)" title="Gemini (chatbot)">Gemini</a> or <a href="Claude_(language_model)" title="Claude (language model)">Claude</a>. LLMs can be <a href="Fine-tuning_(deep_learning)" title="Fine-tuning (deep learning)">fine-tuned</a> for specific tasks or guided by <a href="Prompt_engineering" title="Prompt engineering">prompt engineering</a>.<sup id="cite_ref-few-shot-learners2_1-0" class="reference"><a href="#cite_note-few-shot-learners2-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup> These models acquire <a href="Predictive_learning" title="Predictive learning">predictive power</a> regarding <a href="Syntax" title="Syntax">syntax</a>, <a href="Semantics" title="Semantics">semantics</a>, and <a href="Ontology_(information_science)" title="Ontology (information science)">ontologies</a><sup id="cite_ref-2" class="reference"><a href="#cite_note-2"><span class="cite-bracket">[</span>2<span class="cite-bracket">]</span></a></sup> inherent in human <a href="Text_corpus" title="Text corpus">language corpora</a>, but they also inherit inaccuracies and <a href="Algorithmic_bias" title="Algorithmic bias">biases</a> present in the <a href="Training%2C_validation%2C_and_test_data_sets" title="Training, validation, and test data sets">data</a> they are trained in.<sup id="cite_ref-Manning-2022_3-0" class="reference"><a href="#cite_note-Manning-2022-3"><span class="cite-bracket">[</span>3<span class="cite-bracket">]</span></a></sup>
</p>
<meta property="mw:PageProp/toc">
<div class="mw-heading mw-heading2"><h2 id="History">History</h2></div>


<p>Before the emergence of transformer-based models in 2017, some <a href="Language_model" title="Language model">language models</a> were considered large relative to the computational and data constraints of their time. In the early 1990s, <a href="IBM" title="IBM">IBM</a>'s statistical models pioneered <a href="Bitext_word_alignment" title="Bitext word alignment">word alignment</a> techniques for machine translation, laying the groundwork for <a href="Construction_grammar" title="Construction grammar">corpus-based language modeling</a>. A smoothed <a href="Word_n-gram_language_model" title="Word n-gram language model">n-gram model</a> in 2001, such as those employing <a href="Kneser-Ney_smoothing" class="mw-redirect" title="Kneser-Ney smoothing">Kneser-Ney smoothing</a>, trained on 300 million words achieved state-of-the-art <a href="Perplexity" title="Perplexity">perplexity</a> on benchmark tests at the time.<sup id="cite_ref-4" class="reference"><a href="#cite_note-4"><span class="cite-bracket">[</span>4<span class="cite-bracket">]</span></a></sup> During the 2000s, with the rise of widespread internet access, researchers began compiling massive text datasets from the web ("web as corpus"<sup id="cite_ref-5" class="reference"><a href="#cite_note-5"><span class="cite-bracket">[</span>5<span class="cite-bracket">]</span></a></sup>) to train statistical language models.<sup id="cite_ref-6" class="reference"><a href="#cite_note-6"><span class="cite-bracket">[</span>6<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-7" class="reference"><a href="#cite_note-7"><span class="cite-bracket">[</span>7<span class="cite-bracket">]</span></a></sup>
</p><p>
Moving beyond n-gram models, researchers started in 2000 to use neural networks to learn language models.<sup id="cite_ref-8" class="reference"><a href="#cite_note-8"><span class="cite-bracket">[</span>8<span class="cite-bracket">]</span></a></sup> Following the breakthrough of <a href="Deep_learning" title="Deep learning">deep neural networks</a> in image classification around 2012,<sup id="cite_ref-9" class="reference"><a href="#cite_note-9"><span class="cite-bracket">[</span>9<span class="cite-bracket">]</span></a></sup> similar architectures were adapted for language tasks. This shift was marked by the development of <a href="Word_embedding" title="Word embedding">word embeddings</a> (eg, <a href="Word2vec" title="Word2vec">Word2Vec</a> by Mikolov in 2013) and sequence-to-sequence (<a href="Seq2seq" title="Seq2seq">seq2seq</a>) models using <a href="Long_short-term_memory" title="Long short-term memory">LSTM</a>. In 2016, Google transitioned its translation service to <a href="Neural_machine_translation" title="Neural machine translation">neural machine translation</a> (NMT), replacing statistical phrase-based models with deep <a href="Recurrent_neural_network" title="Recurrent neural network">recurrent neural networks</a>. These early NMT systems used LSTM-based <a href="Encoder-decoder_model" class="mw-redirect" title="Encoder-decoder model">encoder-decoder architectures</a>, as they preceded the invention of <a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">transformers</a>. </p>
<p>At the 2017 <a href="NeurIPS" class="mw-redirect" title="NeurIPS">NeurIPS</a> conference, Google researchers introduced the transformer architecture in their landmark paper "<a href="Attention_Is_All_You_Need" title="Attention Is All You Need">Attention Is All You Need</a>". This paper's goal was to improve upon 2014 seq2seq technology,<sup id="cite_ref-10" class="reference"><a href="#cite_note-10"><span class="cite-bracket">[</span>10<span class="cite-bracket">]</span></a></sup> and was based mainly on the <a href="Attention_(machine_learning)" title="Attention (machine learning)">attention</a> mechanism developed by Bahdanau et al. in 2014.<sup id="cite_ref-11" class="reference"><a href="#cite_note-11"><span class="cite-bracket">[</span>11<span class="cite-bracket">]</span></a></sup> The following year in 2018, <a href="BERT_(language_model)" title="BERT (language model)">BERT</a> was introduced and quickly became "ubiquitous".<sup id="cite_ref-12" class="reference"><a href="#cite_note-12"><span class="cite-bracket">[</span>12<span class="cite-bracket">]</span></a></sup> Though the original transformer has both encoder and decoder blocks, BERT is an encoder-only model. Academic and research usage of BERT began to decline in 2023, following rapid improvements in the abilities of decoder-only models (such as GPT) to solve tasks via <a href="Prompt_engineering" title="Prompt engineering">prompting</a>.<sup id="cite_ref-auto_13-0" class="reference"><a href="#cite_note-auto-13"><span class="cite-bracket">[</span>13<span class="cite-bracket">]</span></a></sup>
</p><p>Although decoder-only <a href="GPT-1" title="GPT-1">GPT-1</a> was introduced in 2018, it was <a href="GPT-2" title="GPT-2">GPT-2</a> in 2019 that caught widespread attention because <a href="OpenAI" title="OpenAI">OpenAI</a> claimed to have initially deemed it too powerful to release publicly, out of fear of malicious use.<sup id="cite_ref-14" class="reference"><a href="#cite_note-14"><span class="cite-bracket">[</span>14<span class="cite-bracket">]</span></a></sup> <a href="GPT-3" title="GPT-3">GPT-3</a> in 2020 went a step further and as of 2025 is available only via <a href="Web_API" title="Web API">API</a> with no offering of downloading the model to execute locally. But it was the 2022 consumer-facing chatbot <a href="ChatGPT" title="ChatGPT">ChatGPT</a> that received extensive media coverage and public attention.<sup id="cite_ref-15" class="reference"><a href="#cite_note-15"><span class="cite-bracket">[</span>15<span class="cite-bracket">]</span></a></sup> The 2023 <a href="GPT-4" title="GPT-4">GPT-4</a> was praised for its increased accuracy and as a "holy grail" for its <a href="Multimodal_learning" title="Multimodal learning">multimodal</a> capabilities.<sup id="cite_ref-16" class="reference"><a href="#cite_note-16"><span class="cite-bracket">[</span>16<span class="cite-bracket">]</span></a></sup> OpenAI did not reveal the high-level architecture and the number of <a href="Parameter#Artificial_intelligence" title="Parameter">parameters</a> of GPT-4. The release of ChatGPT led to an uptick in LLM usage across several research subfields of computer science, including robotics, software engineering, and societal impact work.<sup id="cite_ref-auto_13-1" class="reference"><a href="#cite_note-auto-13"><span class="cite-bracket">[</span>13<span class="cite-bracket">]</span></a></sup> In 2024 OpenAI released the <a href="Reasoning_language_model" title="Reasoning language model">reasoning model</a> <a href="OpenAI_o1" title="OpenAI o1">OpenAI o1</a>, which generates long chains of thought before returning a final answer.<sup id="cite_ref-NYTimesInfo_17-0" class="reference"><a href="#cite_note-NYTimesInfo-17"><span class="cite-bracket">[</span>17<span class="cite-bracket">]</span></a></sup> Many LLMs with parameter counts comparable to those of OpenAI's GPT series have been developed.<sup id="cite_ref-18" class="reference"><a href="#cite_note-18"><span class="cite-bracket">[</span>18<span class="cite-bracket">]</span></a></sup>
</p><p>Since 2022, <a href="Source-available_software" title="Source-available software">source-available</a> models have been gaining popularity, especially at first with <a href="BLOOM_(language_model)" title="BLOOM (language model)">BLOOM</a> and <a href="LLaMA" class="mw-redirect" title="LLaMA">LLaMA</a>, though both have restrictions on the field of use. <a href="Mistral_AI" title="Mistral AI">Mistral AI</a>'s models Mistral 7B and Mixtral 8x7b have the more permissive <a href="Apache_License" title="Apache License">Apache License</a>. In January 2025, <a href="DeepSeek" title="DeepSeek">DeepSeek</a> released DeepSeek R1, a 671-billion-parameter open-weight model that performs comparably to OpenAI o1 but at a much lower cost.<sup id="cite_ref-19" class="reference"><a href="#cite_note-19"><span class="cite-bracket">[</span>19<span class="cite-bracket">]</span></a></sup>
</p><p>Since 2023, many LLMs have been trained to be <a href="Multimodal_learning" title="Multimodal learning">multimodal</a>, having the ability to also process or generate other types of data, such as images or audio. These LLMs are also called large multimodal models (LMMs).<sup id="cite_ref-20" class="reference"><a href="#cite_note-20"><span class="cite-bracket">[</span>20<span class="cite-bracket">]</span></a></sup>
</p><p>As of 2024, the largest and most capable models are all based on the transformer architecture. Some recent implementations are based on other architectures, such as <a href="Recurrent_neural_network" title="Recurrent neural network">recurrent neural network</a> variants and <a href="Mamba_(deep_learning_architecture)" title="Mamba (deep learning architecture)">Mamba</a> (a <a href="State-space_representation" title="State-space representation">state space</a> model).<sup id="cite_ref-21" class="reference"><a href="#cite_note-21"><span class="cite-bracket">[</span>21<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-22" class="reference"><a href="#cite_note-22"><span class="cite-bracket">[</span>22<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-23" class="reference"><a href="#cite_note-23"><span class="cite-bracket">[</span>23<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Dataset_preprocessing">Dataset preprocessing</h2></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="List_of_datasets_for_machine-learning_research#Internet" title="List of datasets for machine-learning research">List of datasets for machine-learning research §&nbsp;Internet</a></div>
<div class="mw-heading mw-heading3"><h3 id="Tokenization">Tokenization</h3></div>
<p>
</p><p>As <a href="Machine_learning" title="Machine learning">machine learning</a> algorithms process numbers rather than text, the text must be converted to numbers. In the first step, a vocabulary is decided upon, then integer indices are arbitrarily but uniquely assigned to each vocabulary entry, and finally, an <a href="Word_embedding" title="Word embedding">embedding</a> is associated to the integer index. Algorithms include <a href="Byte_pair_encoding" class="mw-redirect" title="Byte pair encoding">byte-pair encoding</a> (BPE) and WordPiece. There are also special tokens serving as <a href="Control_character" title="Control character">control characters</a>, such as <code>[MASK]</code> for masked-out token (as used in <a href="BERT_(language_model)" title="BERT (language model)">BERT</a>), and <code>[UNK]</code> ("unknown") for characters not appearing in the vocabulary. Also, some special symbols are used to denote special text formatting. For example, "Ġ" denotes a preceding whitespace in RoBERTa and GPT. "##" denotes continuation of a preceding word in BERT.<sup id="cite_ref-24" class="reference"><a href="#cite_note-24"><span class="cite-bracket">[</span>24<span class="cite-bracket">]</span></a></sup>
</p><p>For example, the BPE tokenizer used by <a href="GPT-3" title="GPT-3">GPT-3</a> (Legacy) would split <small><code>tokenizer: texts -&gt; series of numerical "tokens"</code></small> as
</p>
<table cellpadding="0;" cellspacing="0;" style="border:1px solid black">

<tbody><tr>
<td style="border-left: 2px green; border-right: 2px green">token
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">izer
</td>
<td style="border-left: 2px green; border-right: 2px green">:
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">&nbsp;texts
</td>
<td style="border-left: 2px green; border-right: 2px green">&nbsp;-&gt;
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">series
</td>
<td style="border-left: 2px green; border-right: 2px green">&nbsp;of
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">&nbsp;numerical
</td>
<td style="border-left: 2px green; border-right: 2px green">&nbsp;"
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">t
</td>
<td style="border-left: 2px green; border-right: 2px green">ok
</td>
<td style="background-color: grey; color: white; border-left: 2px green; border-right: 2px green">ens
</td>
<td style="border-left: 2px green; border-right: 2px green">"
</td></tr></tbody></table>
<p>Tokenization also <a href="Data_compression" title="Data compression">compresses</a> the datasets. Because LLMs generally require input to be an <a href="Array_(data_structure)" title="Array (data structure)">array</a> that is not <a href="Jagged_array" title="Jagged array">jagged</a>, the shorter texts must be "padded" until they match the length of the longest one. How many tokens are, on average, needed per word depends on the language of the dataset.<sup id="cite_ref-25" class="reference"><a href="#cite_note-25"><span class="cite-bracket">[</span>25<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-LangModelTokenizsersUnfairness_26-0" class="reference"><a href="#cite_note-LangModelTokenizsersUnfairness-26"><span class="cite-bracket">[</span>26<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="BPE">BPE</h4></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Byte_pair_encoding" class="mw-redirect" title="Byte pair encoding">Byte pair encoding</a></div>
<p>As an example, consider a tokenizer based on byte-pair encoding. In the first step, all unique characters (including blanks and <a href="Punctuation_mark" class="mw-redirect" title="Punctuation mark">punctuation marks</a>) are treated as an initial set of <a href="N-gram" title="N-gram"><i>n</i>-grams</a> (i.e. initial set of uni-grams). Successively the most frequent pair of adjacent characters is merged into a bi-gram and all instances of the pair are replaced by it. All occurrences of adjacent pairs of (previously merged) <i>n</i>-grams that most frequently occur together are then again merged into even lengthier <i>n</i>-gram, until a vocabulary of prescribed size is obtained (in case of GPT-3, the size is 50257).<sup id="cite_ref-xbiWb_27-0" class="reference"><a href="#cite_note-xbiWb-27"><span class="cite-bracket">[</span>27<span class="cite-bracket">]</span></a></sup> After a tokenizer is trained, any text can be tokenized by it, as long as it does not contain characters not appearing in the initial-set of uni-grams.<sup id="cite_ref-2022Book_28-0" class="reference"><a href="#cite_note-2022Book_-28"><span class="cite-bracket">[</span>28<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Problems">Problems</h4></div>
<p>A token vocabulary based on the frequencies extracted from mainly English corpora uses as few tokens as possible for an average English word. However, an average word in another language encoded by such an English-optimized tokenizer is split into a suboptimal amount of tokens. GPT-2 tokenizer can use up to 15 times more tokens per word for some languages, for example for the <a href="Shan_language" title="Shan language">Shan language</a> from <a href="Myanmar" title="Myanmar">Myanmar</a>. Even more widespread languages such as Portuguese and German have "a premium of 50%" compared to English.<sup id="cite_ref-LangModelTokenizsersUnfairness_26-1" class="reference"><a href="#cite_note-LangModelTokenizsersUnfairness-26"><span class="cite-bracket">[</span>26<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Dataset_cleaning">Dataset cleaning</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Data_cleansing" title="Data cleansing">Data cleansing</a></div>
<p>In the context of training LLMs, datasets are typically cleaned by removing low-quality, duplicated, or toxic data.<sup id="cite_ref-aYNg4_29-0" class="reference"><a href="#cite_note-aYNg4-29"><span class="cite-bracket">[</span>29<span class="cite-bracket">]</span></a></sup> Cleaned datasets can increase training efficiency and lead to improved downstream performance.<sup id="cite_ref-30" class="reference"><a href="#cite_note-30"><span class="cite-bracket">[</span>30<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-31" class="reference"><a href="#cite_note-31"><span class="cite-bracket">[</span>31<span class="cite-bracket">]</span></a></sup> A trained LLM can be used to clean datasets for training a further LLM.<sup id="cite_ref-32" class="reference"><a href="#cite_note-32"><span class="cite-bracket">[</span>32<span class="cite-bracket">]</span></a></sup>
</p><p>With the increasing proportion of LLM-generated content on the web, data cleaning in the future may include filtering out such content. LLM-generated content can pose a problem if the content is similar to human text (making filtering difficult) but of lower quality (degrading performance of models trained on it).<sup id="cite_ref-few-shot-learners2_1-1" class="reference"><a href="#cite_note-few-shot-learners2-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Synthetic_data">Synthetic data</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Synthetic_data" title="Synthetic data">Synthetic data</a></div>
<p>Training of largest language models might need more linguistic data than naturally available, or that the naturally occurring data is of insufficient quality. In these cases, synthetic data might be used. Microsoft's Phi series of LLMs is trained on textbook-like data generated by another LLM.<sup id="cite_ref-33" class="reference"><a href="#cite_note-33"><span class="cite-bracket">[</span>33<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Training">Training</h2></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="Fine-tuning_(machine_learning)" class="mw-redirect" title="Fine-tuning (machine learning)">Fine-tuning (machine learning)</a></div>
<p>An LLM is a type of <a href="Foundation_model" title="Foundation model">foundation model</a> (large X model) trained on language. LLMs can be trained in different ways. In particular, GPT models are first pretrained to predict the next word on a large amount of data, before being fine-tuned.
</p>
<div class="mw-heading mw-heading3"><h3 id="Cost">Cost</h3></div>

<p>Substantial infrastructure is necessary for training the largest models. The tendency towards larger models is visible in the <a href="List_of_large_language_models" title="List of large language models">list of large language models</a>. For example, the training of GPT-2 (i.e. a 1.5-billion-parameters model) in 2019 cost $50,000, while training of the PaLM (i.e. a 540-billion-parameters model) in 2022 cost $8 million, and Megatron-Turing NLG 530B (in 2021) cost around $11 million. The qualifier "large" in "large language model" is inherently vague, as there is no definitive threshold for the number of parameters required to qualify as "large". <a href="GPT-1" title="GPT-1">GPT-1</a> of 2018 has 117 million parameters.
</p>
<div class="mw-heading mw-heading3"><h3 id="Fine-tuning">Fine-tuning</h3></div>
<p>Before being <a href="Fine-tuning_(deep_learning)" title="Fine-tuning (deep learning)">fine-tuned</a>, most LLMs are next-token predictors. The fine-tuning adjust the output of an LLM to seem more conversational via techniques like <a href="Reinforcement_learning_from_human_feedback" title="Reinforcement learning from human feedback">reinforcement learning from human feedback</a> (RLHF) or <a href="Constitutional_AI" class="mw-redirect" title="Constitutional AI">constitutional AI</a>.<sup id="cite_ref-34" class="reference"><a href="#cite_note-34"><span class="cite-bracket">[</span>34<span class="cite-bracket">]</span></a></sup>
</p><p>Instruction fine-tuning is a form of <a href="Supervised_learning" title="Supervised learning">supervised learning</a> used to teach LLMs to follow user instructions. In 2022, OpenAI demonstrated InstructGPT, a version of GPT-3 similarly fine-tuned to follow instructions.<sup id="cite_ref-35" class="reference"><a href="#cite_note-35"><span class="cite-bracket">[</span>35<span class="cite-bracket">]</span></a></sup>
</p><p>Reinforcement learning from human feedback (RLHF) involves training a reward model to predict which text humans prefer. Then, the LLM can be fine-tuned through <a href="Reinforcement_learning" title="Reinforcement learning">reinforcement learning</a> to better satisfy this reward model. Since humans typically prefer truthful, helpful and harmless answers, RLHF favors such answers.
</p>
<div class="mw-heading mw-heading2"><h2 id="Architecture">Architecture</h2></div>
<p>LLMs are generally based on the <a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">transformer</a> architecture, which leverages an <a href="Attention_(machine_learning)" title="Attention (machine learning)">attention</a> mechanism that enables the model to process relationships between all elements in a sequence simultaneously, regardless of their distance from each other.<sup id="cite_ref-36" class="reference"><a href="#cite_note-36"><span class="cite-bracket">[</span>36<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Attention_mechanism_and_context_window">Attention mechanism and context window</h3></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="Attention_(machine_learning)" title="Attention (machine learning)">Attention (machine learning)</a></div>

<p>In order to find out which tokens are relevant to each other within the scope of the context window, the attention mechanism calculates "soft" weights for each token, more precisely for its embedding, by using multiple attention heads, each with its own "relevance" for calculating its own soft weights. For example, the small (i.e. 117M parameter sized) <a href="GPT-2" title="GPT-2">GPT-2</a> model has had twelve attention heads and a context window of only 1k tokens.<sup id="cite_ref-Jay_Allamar_GPT2_38-0" class="reference"><a href="#cite_note-Jay_Allamar_GPT2-38"><span class="cite-bracket">[</span>38<span class="cite-bracket">]</span></a></sup> In its medium version it has 345M parameters and contains 24 layers, each with 12 attention heads. For the training with gradient descent a batch size of 512 was utilized.<sup id="cite_ref-2022Book_28-1" class="reference"><a href="#cite_note-2022Book_-28"><span class="cite-bracket">[</span>28<span class="cite-bracket">]</span></a></sup>
</p><p>The largest models, such as Google's <a href="Gemini_(language_model)" title="Gemini (language model)">Gemini 1.5</a>, presented in February 2024, can have a context window sized up to 1 million (context window of 10 million was also "successfully tested").<sup id="cite_ref-39" class="reference"><a href="#cite_note-39"><span class="cite-bracket">[</span>39<span class="cite-bracket">]</span></a></sup> Other models with large context windows includes Anthropic's Claude 2.1, with a context window of up to 200k tokens.<sup id="cite_ref-40" class="reference"><a href="#cite_note-40"><span class="cite-bracket">[</span>40<span class="cite-bracket">]</span></a></sup> Note that this maximum refers to the number of input tokens and that the maximum number of output tokens differs from the input and is often smaller. For example, the GPT-4 Turbo model has a maximum output of 4096 tokens.<sup id="cite_ref-41" class="reference"><a href="#cite_note-41"><span class="cite-bracket">[</span>41<span class="cite-bracket">]</span></a></sup>
</p><p>Length of a conversation that the model can take into account when generating its next answer is limited by the size of a context window, as well. If the length of a conversation, for example with <a href="ChatGPT" title="ChatGPT">ChatGPT</a>, is longer than its context window, only the parts inside the context window are taken into account when generating the next answer, or the model needs to apply some algorithm to summarize the too distant parts of conversation.
</p><p>The shortcomings of making a context window larger include higher computational cost and possibly diluting the focus on local context, while making it smaller can cause a model to miss an important long-range dependency. Balancing them is a matter of experimentation and domain-specific considerations.
</p><p>A model may be pre-trained either to predict how the segment continues, or what is missing in the segment, given a segment from its training dataset.<sup id="cite_ref-ioUpE_42-0" class="reference"><a href="#cite_note-ioUpE-42"><span class="cite-bracket">[</span>42<span class="cite-bracket">]</span></a></sup> It can be either
</p>
<ul><li>autoregressive (i.e. predicting how the segment continues, as <a href="Generative_pretrained_transformer" class="mw-redirect" title="Generative pretrained transformer">GPTs</a> do): for example given a segment "I like to eat", the model predicts "ice cream", or "sushi".</li>
<li>"<a href="Cloze_test" title="Cloze test">masked</a>" (i.e. filling in the parts missing from the segment, the way "BERT"<sup id="cite_ref-jm_43-0" class="reference"><a href="#cite_note-jm-43"><span class="cite-bracket">[</span>43<span class="cite-bracket">]</span></a></sup> does it): for example, given a segment "I like to <code>[__] [__]</code> cream", the model predicts that "eat" and "ice" are missing.</li></ul>
<p>Models may be trained on auxiliary tasks which test their understanding of the data distribution, such as Next Sentence Prediction (NSP), in which pairs of sentences are presented and the model must predict whether they appear consecutively in the training corpus.<sup id="cite_ref-jm_43-1" class="reference"><a href="#cite_note-jm-43"><span class="cite-bracket">[</span>43<span class="cite-bracket">]</span></a></sup> During training, <a href="Regularization_(mathematics)" title="Regularization (mathematics)">regularization</a> loss is also used to stabilize training. However regularization loss is usually not used during <a href="Training%2C_validation%2C_and_test_data_sets" title="Training, validation, and test data sets">testing</a> and evaluation.
</p>
<div class="mw-heading mw-heading3"><h3 id="Mixture_of_experts">Mixture of experts</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Mixture_of_experts" title="Mixture of experts">Mixture of experts</a></div>
<p>A <a href="Mixture_of_experts" title="Mixture of experts">mixture of experts</a> (MoE) is a <a href="Machine_learning" title="Machine learning">machine learning</a> architecture in which multiple specialized neural networks ("experts") work together, with a gating mechanism that routes each input to the most appropriate expert(s). Mixtures of experts can reduce inference costs, as only a fraction of the parameters are used for each input. The approach was introduced in 2017 by Google researchers.<sup id="cite_ref-HGZCJ_44-0" class="reference"><a href="#cite_note-HGZCJ-44"><span class="cite-bracket">[</span>44<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-R9Qq5_45-0" class="reference"><a href="#cite_note-R9Qq5-45"><span class="cite-bracket">[</span>45<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-glam-blog_46-0" class="reference"><a href="#cite_note-glam-blog-46"><span class="cite-bracket">[</span>46<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Parameter_size">Parameter size</h3></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="1.58-bit_large_language_model" title="1.58-bit large language model">1.58-bit large language model</a></div>
<p>Typically, LLMs are trained with single- or half-precision <a href="Floating_point_numbers" class="mw-redirect" title="Floating point numbers">floating point numbers</a> (float32 and float16). One float16 has 16 bits, or 2 bytes, and so one billion parameters require 2 gigabytes. The largest models typically have 100 billion parameters, requiring 200 gigabytes to load, which places them outside the range of most consumer electronics.<sup id="cite_ref-47" class="reference"><a href="#cite_note-47"><span class="cite-bracket">[</span>47<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Quantization">Quantization</h4></div>
<p><i>Post-training <a href="Quantization_(signal_processing)" title="Quantization (signal processing)">quantization</a></i><sup id="cite_ref-LS2Go_48-0" class="reference"><a href="#cite_note-LS2Go-48"><span class="cite-bracket">[</span>48<span class="cite-bracket">]</span></a></sup> aims to decrease the space requirement by lowering precision of the parameters of a trained model, while preserving most of its performance. Quantization can be further classified as <i>static quantization</i> if the quantization parameters are determined beforehand (typically during a calibration phase), and <i>dynamic quantization</i> if the quantization is applied during inference. The simplest form of quantization simply truncates all the parameters to a given number of bits: this is applicable to static as well as dynamic quantization, but loses much precision. Dynamic quantization allows for the use of a different quantization <a href="Codebook#Data_compression" title="Codebook">codebook</a> per layer, either a lookup table of values or a linear mapping (scaling factor and bias), at the cost of foregoing the possible speed improvements from using lower-precision arithmetic.
</p><p>Quantized models are typically seen as frozen with modification of weights (e.g. fine-tuning) only applied to the original model. It is possible to fine-tune quantized models using <a href="LoRA" class="mw-redirect" title="LoRA">low-rank adaptation</a>.
</p>
<div class="mw-heading mw-heading2"><h2 id="Extensibility">Extensibility</h2></div>
<p>Beyond basic text generation, various techniques have been developed to extend LLM capabilities, including the use of external tools and data sources, improved reasoning on complex problems, and enhanced instruction-following or autonomy through prompting methods.
</p>
<div class="mw-heading mw-heading3"><h3 id="Prompt_engineering">Prompt engineering</h3></div>
<p>In 2020, <a href="OpenAI" title="OpenAI">OpenAI</a> researchers demonstrated that their new model <a href="GPT-3" title="GPT-3">GPT-3</a> could understand what format to use given a few rounds of Q and A (or other type of task) in the input data as example, thanks in part due to the RLHF technique. This technique, called <i>few-shot prompting</i>, allows LLMs to be adapted to any task without requiring fine-tuning.<sup id="cite_ref-few-shot-learners2_1-2" class="reference"><a href="#cite_note-few-shot-learners2-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup> Also in 2022, it was found that the base GPT-3 model can generate an instruction based on user input. The generated instruction along with user input is then used as input to another instance of the model under a "Instruction: [...], Input: [...], Output:" format. The other instance is able to complete the output and often produces the correct answer in doing so. The ability to "self-instruct" makes LLMs able to <a href="Bootstrapping" title="Bootstrapping">bootstrap</a> themselves toward a correct answer.<sup id="cite_ref-self-instruct-paper_49-0" class="reference"><a href="#cite_note-self-instruct-paper-49"><span class="cite-bracket">[</span>49<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Dialogue_processing_(chatbot)">Dialogue processing (chatbot)</h3></div>
<p>An LLM can be turned into a chatbot or a "dialog assistant" by specializing it for conversation. In essence, user input is prefixed with a marker such as "Q:" or "User:" and the LLM is asked to predict the output after a fixed "A:" or "Assistant:". This type of model became commercially available in 2022 with ChatGPT, a sibling model of InstructGPT fine-tuned to accept and produce dialog-formatted text based on GPT-3.5. It could similarly follow user instructions.<sup id="cite_ref-50" class="reference"><a href="#cite_note-50"><span class="cite-bracket">[</span>50<span class="cite-bracket">]</span></a></sup> Before the stream of User and Assistant lines, a chat context usually start with a few lines of overarching instructions, from a role called "developer" or "system" to convey a higher authority than the user's input. This is called a "system prompt".<sup id="cite_ref-51" class="reference"><a href="#cite_note-51"><span class="cite-bracket">[</span>51<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-52" class="reference"><a href="#cite_note-52"><span class="cite-bracket">[</span>52<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Retrieval-augmented_generation">Retrieval-augmented generation</h3></div>
<p><a href="Retrieval-augmented_generation" title="Retrieval-augmented generation">Retrieval-augmented generation</a> (RAG) is an approach that enhances LLMs by integrating them with <a href="Document_retrieval" title="Document retrieval">document retrieval</a> systems. Given a query, a document retriever is called to retrieve the most relevant documents. This is usually done by encoding the query and the documents into vectors, then finding the documents with vectors (usually stored in a <a href="Vector_database" title="Vector database">vector database</a>) most similar to the vector of the query. The LLM then generates an output based on both the query and context included from the retrieved documents.<sup id="cite_ref-BUZBP_53-0" class="reference"><a href="#cite_note-BUZBP-53"><span class="cite-bracket">[</span>53<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Tool_use">Tool use</h3></div>
<p>Tool use is a mechanism that enables LLMs to interact with external systems, applications, or data sources. It can allow for example to fetch real-time information from an API or to execute code. A program separate from the LLM watches the output stream of the LLM for a special tool-calling syntax. When these special tokens appear, the program calls the tool accordingly and feeds its output back into the LLM's input stream.<sup id="cite_ref-54" class="reference"><a href="#cite_note-54"><span class="cite-bracket">[</span>54<span class="cite-bracket">]</span></a></sup>
</p><p>Early tool-using LLMs were fine-tuned on the use of specific tools. But fine-tuning LLMs for the ability to read <a href="API" title="API">API</a> documentation and call API correctly has greatly expanded the range of tools accessible to an LLM.<sup id="cite_ref-lLrda_55-0" class="reference"><a href="#cite_note-lLrda-55"><span class="cite-bracket">[</span>55<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-4Xzrs_56-0" class="reference"><a href="#cite_note-4Xzrs-56"><span class="cite-bracket">[</span>56<span class="cite-bracket">]</span></a></sup> Describing available tools in the system prompt can also make an LLM able to use tools. A system prompt instructing ChatGPT (GPT-4) to use multiple types of tools can be found online.<sup id="cite_ref-57" class="reference"><a href="#cite_note-57"><span class="cite-bracket">[</span>57<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Agency">Agency</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="AI_agent" class="mw-redirect" title="AI agent">AI agent</a></div>
<p>An LLM is typically not an <a href="Autonomous_agent" title="Autonomous agent">autonomous agent</a> by itself, as it lacks the ability to interact with dynamic environments, recall past behaviors, and plan future actions. But it can be transformed into an agent by adding supporting elements: the role (profile) and the surrounding environment of an agent can be additional inputs to the LLM, while memory can be integrated as a tool or provided as additional input. Instructions and input patterns are used to make the LLM plan actions and tool use is used to potentially carry out these actions.<sup id="cite_ref-58" class="reference"><a href="#cite_note-58"><span class="cite-bracket">[</span>58<span class="cite-bracket">]</span></a></sup>
</p><p>The ReAct pattern, a portmanteau of "Reason&nbsp;+&nbsp;Act", constructs an <a href="Intelligent_agent" title="Intelligent agent">agent</a> out of an LLM, using the LLM as a planner. The LLM is prompted to "think out loud". Specifically, the language model is prompted with a textual description of the environment, a goal, a list of possible actions, and a record of the actions and observations so far. It generates one or more thoughts before generating an action, which is then executed in the environment.<sup id="cite_ref-DmvNE_59-0" class="reference"><a href="#cite_note-DmvNE-59"><span class="cite-bracket">[</span>59<span class="cite-bracket">]</span></a></sup>
</p><p>In the DEPS ("Describe, Explain, Plan and Select") method, an LLM is first connected to the visual world via image descriptions. It is then prompted to produce plans for complex tasks and behaviors based on its pretrained knowledge and the environmental feedback it receives.<sup id="cite_ref-60" class="reference"><a href="#cite_note-60"><span class="cite-bracket">[</span>60<span class="cite-bracket">]</span></a></sup>
</p><p>The Reflexion method<sup id="cite_ref-sbB2T_61-0" class="reference"><a href="#cite_note-sbB2T-61"><span class="cite-bracket">[</span>61<span class="cite-bracket">]</span></a></sup> constructs an agent that learns over multiple episodes. At the end of each episode, the LLM is given the record of the episode, and prompted to think up "lessons learned", which would help it perform better at a subsequent episode. These "lessons learned" are stored as a form of long-term memory and given to the agent in the subsequent episodes.<sup id="cite_ref-sbB2T_61-1" class="reference"><a href="#cite_note-sbB2T-61"><span class="cite-bracket">[</span>61<span class="cite-bracket">]</span></a></sup>
</p><p><a href="Monte_Carlo_tree_search" title="Monte Carlo tree search">Monte Carlo tree search</a> can use an LLM as rollout heuristic. When a programmatic world model is not available, an LLM can also be prompted with a description of the environment to act as world model.<sup id="cite_ref-ltTer_62-0" class="reference"><a href="#cite_note-ltTer-62"><span class="cite-bracket">[</span>62<span class="cite-bracket">]</span></a></sup>
</p><p>For open-ended exploration, an LLM can be used to score observations for their "interestingness", which can be used as a reward signal to guide a normal (non-LLM) reinforcement learning agent.<sup id="cite_ref-mBvD9_63-0" class="reference"><a href="#cite_note-mBvD9-63"><span class="cite-bracket">[</span>63<span class="cite-bracket">]</span></a></sup> Alternatively, it can <a href="Zone_of_proximal_development" title="Zone of proximal development">propose increasingly difficult tasks</a> for <a href="Curriculum_learning" title="Curriculum learning">curriculum learning</a>.<sup id="cite_ref-:0_64-0" class="reference"><a href="#cite_note-:0-64"><span class="cite-bracket">[</span>64<span class="cite-bracket">]</span></a></sup> Instead of outputting individual actions, an LLM planner can also construct "skills", or <a href="Function_(computer_programming)" title="Function (computer programming)">functions</a> for complex action sequences. The skills can be stored and later invoked, allowing increasing levels of abstraction in planning.<sup id="cite_ref-:0_64-1" class="reference"><a href="#cite_note-:0-64"><span class="cite-bracket">[</span>64<span class="cite-bracket">]</span></a></sup>
</p><p>Multiple agents with memory can interact socially.<sup id="cite_ref-XuvjF_65-0" class="reference"><a href="#cite_note-XuvjF-65"><span class="cite-bracket">[</span>65<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Reasoning">Reasoning</h3></div>
<p>LLMs are conventionally trained to generate an output without generating intermediate steps. As a result their performance tends to be subpar on complex questions requiring (at least in humans) intermediate steps of thought. This deficiency has been overcome by breaking down the tasks into smaller steps for the LLM either manually or automatically.
</p>
<div class="mw-heading mw-heading4"><h4 id="Chaining">Chaining</h4></div>
<p>The "prompt chaining" paradigm was published in 2021.<sup id="cite_ref-auto2_66-0" class="reference"><a href="#cite_note-auto2-66"><span class="cite-bracket">[</span>66<span class="cite-bracket">]</span></a></sup> In this method, a user manually breaks a complex problem down into several steps. In each step, the LLM receives as input a prompt telling it what to do and some results from preceeding steps. The result from one step is then reused in a next step, until a final answer is reached. The ability of an LLM to follow instructions means that even non-experts can write a successful collection of step-wise prompts given a few rounds of trial and error.<sup id="cite_ref-67" class="reference"><a href="#cite_note-67"><span class="cite-bracket">[</span>67<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-68" class="reference"><a href="#cite_note-68"><span class="cite-bracket">[</span>68<span class="cite-bracket">]</span></a></sup>
</p><p>A 2022 paper demonstrated a separate technique called "<a href="Chain-of-thought_prompting" class="mw-redirect" title="Chain-of-thought prompting">Chain-of-Thought</a> Prompting", which makes the LLM break the question down autonomously. An LLM is given some examples where the "assistant" verbally breaks down the thought process before arriving at an answer. The LLM mimics these examples and also tries to spend some time generating intermediate steps before providing the final answer. This additional step elicited by prompting improves the correctness of the LLM on relatively complex questions. On math word questions, a prompted model can exceed even fine-tuned GPT-3 with a verifier.<sup id="cite_ref-auto2_66-1" class="reference"><a href="#cite_note-auto2-66"><span class="cite-bracket">[</span>66<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-69" class="reference"><a href="#cite_note-69"><span class="cite-bracket">[</span>69<span class="cite-bracket">]</span></a></sup> Chain-of-thought can also be elicited by simply adding an instruction like "Let's think step by step" to the prompt, in order to encourage the LLM to proceed methodically instead of trying to directly guess the answer.<sup id="cite_ref-70" class="reference"><a href="#cite_note-70"><span class="cite-bracket">[</span>70<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Model-native_reasoning">Model-native reasoning</h4></div>
<div role="note" class="hatnote navigation-not-searchable">Main articles: <a href="Reasoning_language_model" title="Reasoning language model">Reasoning language model</a> and <a href="Reflection_(artificial_intelligence)" class="mw-redirect" title="Reflection (artificial intelligence)">Reflection (artificial intelligence)</a></div>
<p>In late 2024 "reasoning models" were released. These were trained to spend more time generating step-by-step solutions before providing final answers, which ws intended to be similar to human problem-solving processes. OpenAI introduced this concept with their <a href="OpenAI_o1" title="OpenAI o1">o1</a> model in September 2024, followed by <a href="OpenAI_o3" title="OpenAI o3">o3</a> in April 2025. On the <a href="International_Mathematical_Olympiad" title="International Mathematical Olympiad">International Mathematics Olympiad</a> qualifying exam problems, <a href="GPT-4o" title="GPT-4o">GPT-4o</a> achieved 13% accuracy while o1 reached 83%.<sup id="cite_ref-nyt-o3_71-0" class="reference"><a href="#cite_note-nyt-o3-71"><span class="cite-bracket">[</span>71<span class="cite-bracket">]</span></a></sup>
</p><p>In January 2025, the Chinese company DeepSeek released DeepSeek-R1, a 671-billion-parameter open-weight reasoning model that achieved comparable performance to OpenAI's o1 while being significantly more cost-effective to operate. Unlike proprietary models from OpenAI, DeepSeek-R1's open-weight nature allowed researchers to study and build upon the algorithm, though its training data remained private.<sup id="cite_ref-nature-deepseek_72-0" class="reference"><a href="#cite_note-nature-deepseek-72"><span class="cite-bracket">[</span>72<span class="cite-bracket">]</span></a></sup>
</p><p>These reasoning models typically require more computational resources per query compared to traditional LLMs, as they perform more extensive processing to work through problems step-by-step.<sup id="cite_ref-nyt-o3_71-1" class="reference"><a href="#cite_note-nyt-o3-71"><span class="cite-bracket">[</span>71<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Inference_optimization">Inference optimization</h3></div>
<p>Inference optimization refers to techniques that improve LLM performance by applying additional computational resources during the inference process, rather than requiring model retraining. These approaches implement various state-of-the-art reasoning and decision-making strategies to enhance accuracy and capabilities.
</p><p><b>OptiLLM</b> is an <a href="OpenAI" title="OpenAI">OpenAI</a> API-compatible optimizing inference proxy that implements multiple inference optimization techniques simultaneously.<sup id="cite_ref-73" class="reference"><a href="#cite_note-73"><span class="cite-bracket">[</span>73<span class="cite-bracket">]</span></a></sup> The system acts as a transparent proxy that can work with any LLM provider, implementing techniques such as <a href="Monte_Carlo_tree_search" title="Monte Carlo tree search">Monte Carlo tree search</a> (MCTS), <a href="Mixture_of_experts" title="Mixture of experts">mixture of agents</a> (MOA), best-of-N sampling, and chain-of-thought reflection. OptiLLM demonstrates that strategic application of computational resources at inference time can substantially improve model performance across diverse tasks, achieving significant improvements on benchmarks such as the AIME 2024 mathematics competition and various coding challenges.<sup id="cite_ref-74" class="reference"><a href="#cite_note-74"><span class="cite-bracket">[</span>74<span class="cite-bracket">]</span></a></sup>
</p><p>These inference optimization approaches represent a growing category of tools that enhance existing LLMs without requiring access to model weights or retraining, making advanced reasoning capabilities more accessible across different model providers and use cases.
</p>
<div class="mw-heading mw-heading2"><h2 id="Forms_of_input_and_output">Forms of input and output</h2></div>
<div class="mw-heading mw-heading3"><h3 id="Multimodality">Multimodality</h3></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="Multimodal_learning" title="Multimodal learning">Multimodal learning</a></div>
<p>Multimodality means having multiple modalities, where a "<a href="Modality_(human%E2%80%93computer_interaction)" title="Modality (human–computer interaction)">modality</a>" refers to a type of input or output, such as video, image, audio, text, <a href="Proprioception" title="Proprioception">proprioception</a>, etc.<sup id="cite_ref-75" class="reference"><a href="#cite_note-75"><span class="cite-bracket">[</span>75<span class="cite-bracket">]</span></a></sup> For example, <a href="Pathways_Language_Model" class="mw-redirect" title="Pathways Language Model">Google PaLM</a> model was fine-tuned into a multimodal model and applied to <a href="Robot_control" title="Robot control">robotic control</a>.<sup id="cite_ref-76" class="reference"><a href="#cite_note-76"><span class="cite-bracket">[</span>76<span class="cite-bracket">]</span></a></sup> <a href="LLaMA" class="mw-redirect" title="LLaMA">LLaMA</a> models have also been turned multimodal using the tokenization method, to allow image inputs,<sup id="cite_ref-77" class="reference"><a href="#cite_note-77"><span class="cite-bracket">[</span>77<span class="cite-bracket">]</span></a></sup> and video inputs.<sup id="cite_ref-78" class="reference"><a href="#cite_note-78"><span class="cite-bracket">[</span>78<span class="cite-bracket">]</span></a></sup> <a href="GPT-4o" title="GPT-4o">GPT-4o</a> can process and generate text, audio and images.<sup id="cite_ref-79" class="reference"><a href="#cite_note-79"><span class="cite-bracket">[</span>79<span class="cite-bracket">]</span></a></sup> Such models are sometimes called large multimodal models (LMMs).<sup id="cite_ref-80" class="reference"><a href="#cite_note-80"><span class="cite-bracket">[</span>80<span class="cite-bracket">]</span></a></sup>
</p><p>A common method to create multimodal models out of an LLM is to "tokenize" the output of a trained encoder. Concretely, one can construct an LLM that can understand images as follows: take a trained LLM, and take a trained image encoder <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle E}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>E</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle E}</annotation>
</semantics>
</math></span><img src="./4232c9de2ee3eec0a9c0a19b15ab92daa6223f9b.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.776ex; height:2.176ex;" alt="{\displaystyle E}" loading="lazy"></span>. Make a small multilayered perceptron <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f}</annotation>
</semantics>
</math></span><img src="./132e57acb643253e7810ee9702d9581f159a1c61.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:1.279ex; height:2.509ex;" alt="{\displaystyle f}" loading="lazy"></span>, so that for any image <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y}</annotation>
</semantics>
</math></span><img src="./b8a6208ec717213d4317e666f1ae872e00620a0d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:1.155ex; height:2.009ex;" alt="{\displaystyle y}" loading="lazy"></span>, the post-processed vector <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(E(y))}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>E</mi>
<mo stretchy="false">(</mo>
<mi>y</mi>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(E(y))}</annotation>
</semantics>
</math></span><img src="./8d41d0ec0611a795f65ea14a43b8016462703a8e.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:7.828ex; height:2.843ex;" alt="{\displaystyle f(E(y))}" loading="lazy"></span> has the same dimensions as an encoded token. That is an "image token". Then, one can interleave text tokens and image tokens. The compound model is then fine-tuned on an image-text dataset. This basic construction can be applied with more sophistication to improve the model. The image encoder may be frozen to improve stability.<sup id="cite_ref-81" class="reference"><a href="#cite_note-81"><span class="cite-bracket">[</span>81<span class="cite-bracket">]</span></a></sup> The model Flamingo demonstrated in 2022 the effectiveness of the tokenization method, fine-tuning a pair of pretrained language model and image encoder to perform better on visual question answering than models trained from scratch.<sup id="cite_ref-82" class="reference"><a href="#cite_note-82"><span class="cite-bracket">[</span>82<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Non-natural_languages">Non-natural languages</h3></div>
<p>LLMs can handle programming languages similarly to how they handle natural languages. No special change in token handling is needed as code, like human language, is represented as plain text. LLMs can generate code based on problems or instructions written in <a href="Natural_language" title="Natural language">natural language</a>. They can also describe code in natural language or translate between programming languages. They were originally used as a <a href="Code_completion" title="Code completion">code completion</a> tool, but advances have moved them towards <a href="Automatic_programming" title="Automatic programming">automatic programming</a>. Services such as <a href="GitHub_Copilot" title="GitHub Copilot">GitHub Copilot</a> offer LLMs specifically trained, fine-tuned, or prompted for programming.<sup id="cite_ref-83" class="reference"><a href="#cite_note-83"><span class="cite-bracket">[</span>83<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-84" class="reference"><a href="#cite_note-84"><span class="cite-bracket">[</span>84<span class="cite-bracket">]</span></a></sup>
</p><p>LLM architectures have also proven useful in analyzing biological sequences: protein, DNA, and RNA. With proteins they appear able to capture a degree of "grammar" from the amino-acid sequence, condensing a sequence into an <a href="Embedding_(machine_learning)" title="Embedding (machine learning)">embedding</a>. On tasks such as structure prediction and mutational outcome prediction, a small model using an embedding as input can approach or exceed much larger models using <a href="Multiple_sequence_alignment" title="Multiple sequence alignment">multiple sequence alignments</a> (MSA) as input.<sup id="cite_ref-85" class="reference"><a href="#cite_note-85"><span class="cite-bracket">[</span>85<span class="cite-bracket">]</span></a></sup> ESMFold, <a href="Meta_Platforms" title="Meta Platforms">Meta Platforms</a>' embedding-based method for protein structure prediction, runs an order of magnitude faster than <a href="AlphaFold2" class="mw-redirect" title="AlphaFold2">AlphaFold2</a> thanks to the removal of an MSA requirement and a lower parameter count due to the use of embeddings.<sup id="cite_ref-86" class="reference"><a href="#cite_note-86"><span class="cite-bracket">[</span>86<span class="cite-bracket">]</span></a></sup> Meta hosts ESM Atlas, a database of 772 million structures of <a href="Metagenomic" class="mw-redirect" title="Metagenomic">metagenomic</a> proteins predicted using ESMFold.<sup id="cite_ref-87" class="reference"><a href="#cite_note-87"><span class="cite-bracket">[</span>87<span class="cite-bracket">]</span></a></sup> An LLM can also design proteins unlike any seen in nature.<sup id="cite_ref-88" class="reference"><a href="#cite_note-88"><span class="cite-bracket">[</span>88<span class="cite-bracket">]</span></a></sup> Nucleic acid models have proven useful in detecting <a href="Regulatory_sequence" title="Regulatory sequence">regulatory sequences</a>,<sup id="cite_ref-89" class="reference"><a href="#cite_note-89"><span class="cite-bracket">[</span>89<span class="cite-bracket">]</span></a></sup> sequence classification, RNA-RNA interaction prediction, and RNA structure prediction.<sup id="cite_ref-90" class="reference"><a href="#cite_note-90"><span class="cite-bracket">[</span>90<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Properties">Properties</h2></div>
<div class="mw-heading mw-heading3"><h3 id="Scaling_laws">Scaling laws</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Neural_scaling_law" title="Neural scaling law">Neural scaling law</a></div>
<p>The performance of an LLM after pretraining largely depends on the:
</p>
<ul><li>cost of pretraining <small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle C}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>C</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle C}</annotation>
</semantics>
</math></span><img src="./4fc55753007cd3c18576f7933f6f089196732029.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.766ex; height:2.176ex;" alt="{\displaystyle C}" loading="lazy"></span></small> (the total amount of compute used),</li>
<li>size of the <a href="Artificial_neural_network" class="mw-redirect" title="Artificial neural network">artificial neural network</a> itself, such as number of parameters <small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle N}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>N</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle N}</annotation>
</semantics>
</math></span><img src="./f5e3890c981ae85503089652feb48b191b57aae3.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:2.064ex; height:2.176ex;" alt="{\displaystyle N}" loading="lazy"></span></small> (i.e. amount of neurons in its layers, amount of weights between them and biases),</li>
<li>size of its pretraining dataset (i.e. number of tokens in corpus, <small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle D}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>D</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle D}</annotation>
</semantics>
</math></span><img src="./f34a0c600395e5d4345287e21fb26efd386990e6.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.924ex; height:2.176ex;" alt="{\displaystyle D}" loading="lazy"></span></small>).</li></ul>
<p>"Scaling laws" are <a href="Empirical_statistical_laws" title="Empirical statistical laws">empirical statistical laws</a> that predict LLM performance based on such factors. One particular scaling law ("<a href="Chinchilla_AI" class="mw-redirect" title="Chinchilla AI">Chinchilla scaling</a>") for LLM autoregressively trained for one epoch, with a <a href="Log-log_plot" class="mw-redirect" title="Log-log plot">log-log</a> <a href="Learning_rate" title="Learning rate">learning rate</a> schedule, states that:<sup id="cite_ref-fJta3_91-0" class="reference"><a href="#cite_note-fJta3-91"><span class="cite-bracket">[</span>91<span class="cite-bracket">]</span></a></sup>
<span class="mwe-math-element mwe-math-element-block"><span class="mwe-math-mathml-display mwe-math-mathml-a11y" style="display: none;"><math display="block" xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{cases}C=C_{0}ND\\[6pt]L={\frac {A}{N^{\alpha }}}+{\frac {B}{D^{\beta }}}+L_{0}\end{cases}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mrow>
<mo>{</mo>
<mtable columnalign="left left" rowspacing="0.8em 0.2em" columnspacing="1em" displaystyle="false">
<mtr>
<mtd>
<mi>C</mi>
<mo>=</mo>
<msub>
<mi>C</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>0</mn>
</mrow>
</msub>
<mi>N</mi>
<mi>D</mi>
</mtd>
</mtr>
<mtr>
<mtd>
<mi>L</mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mi>A</mi>
<msup>
<mi>N</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>α<!-- α --></mi>
</mrow>
</msup>
</mfrac>
</mrow>
<mo>+</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mi>B</mi>
<msup>
<mi>D</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>β<!-- β --></mi>
</mrow>
</msup>
</mfrac>
</mrow>
<mo>+</mo>
<msub>
<mi>L</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>0</mn>
</mrow>
</msub>
</mtd>
</mtr>
</mtable>
<mo fence="true" stretchy="true" symmetric="true"></mo>
</mrow>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{cases}C=C_{0}ND\\[6pt]L={\frac {A}{N^{\alpha }}}+{\frac {B}{D^{\beta }}}+L_{0}\end{cases}}}</annotation>
</semantics>
</math></span></span> where the variables are
</p>
<ul><li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle C}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>C</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle C}</annotation>
</semantics>
</math></span><img src="./4fc55753007cd3c18576f7933f6f089196732029.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.766ex; height:2.176ex;" alt="{\displaystyle C}" loading="lazy"></span></small> is the cost of training the model, in <a href="FLOPS" class="mw-redirect" title="FLOPS">FLOPs</a>.</li>
<li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle N}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>N</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle N}</annotation>
</semantics>
</math></span><img src="./f5e3890c981ae85503089652feb48b191b57aae3.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:2.064ex; height:2.176ex;" alt="{\displaystyle N}" loading="lazy"></span></small> is the number of parameters in the model.</li>
<li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle D}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>D</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle D}</annotation>
</semantics>
</math></span><img src="./f34a0c600395e5d4345287e21fb26efd386990e6.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.924ex; height:2.176ex;" alt="{\displaystyle D}" loading="lazy"></span></small> is the number of tokens in the training set.</li>
<li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle L}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>L</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle L}</annotation>
</semantics>
</math></span><img src="./103168b86f781fe6e9a4a87b8ea1cebe0ad4ede8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.583ex; height:2.176ex;" alt="{\displaystyle L}" loading="lazy"></span></small> is the average negative log-likelihood loss per token (<a href="Nat_(unit)" title="Nat (unit)">nats</a>/token), achieved by the trained LLM on the test dataset.</li></ul>
<p>and the statistical hyper-parameters are
</p>
<ul><li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle C_{0}=6}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>C</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>0</mn>
</mrow>
</msub>
<mo>=</mo>
<mn>6</mn>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle C_{0}=6}</annotation>
</semantics>
</math></span><img src="./b05c98b1743f05e046a3f3bb0a966fa898e431e2.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:6.977ex; height:2.509ex;" alt="{\displaystyle C_{0}=6}" loading="lazy"></span></small>, meaning that it costs 6 FLOPs per parameter to train on one token. Note that training cost is much higher than inference cost, where it costs 1 to 2 FLOPs per parameter to infer on one token.</li>
<li><small><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \alpha =0.34,\beta =0.28,A=406.4,B=410.7,L_{0}=1.69}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>α<!-- α --></mi>
<mo>=</mo>
<mn>0.34</mn>
<mo>,</mo>
<mi>β<!-- β --></mi>
<mo>=</mo>
<mn>0.28</mn>
<mo>,</mo>
<mi>A</mi>
<mo>=</mo>
<mn>406.4</mn>
<mo>,</mo>
<mi>B</mi>
<mo>=</mo>
<mn>410.7</mn>
<mo>,</mo>
<msub>
<mi>L</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>0</mn>
</mrow>
</msub>
<mo>=</mo>
<mn>1.69</mn>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \alpha =0.34,\beta =0.28,A=406.4,B=410.7,L_{0}=1.69}</annotation>
</semantics>
</math></span><img src="./848b6d78d881ed6da8d6b60e8d788bc799525401.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:51.588ex; height:2.509ex;" alt="{\displaystyle \alpha =0.34,\beta =0.28,A=406.4,B=410.7,L_{0}=1.69}" loading="lazy"></span></small></li></ul>
<div class="mw-heading mw-heading3"><h3 id="Emergent_abilities">Emergent abilities</h3></div>
<p></p>
<p>Performance of bigger models on various tasks, when plotted on a log-log scale, appears as a linear extrapolation of performance achieved by smaller models. However, this linearity may be punctuated by "<a href="Broken_Neural_Scaling_Law" class="mw-redirect" title="Broken Neural Scaling Law">break(s)</a>"<sup id="cite_ref-IYm4Q_92-1" class="reference"><a href="#cite_note-IYm4Q-92"><span class="cite-bracket">[</span>92<span class="cite-bracket">]</span></a></sup> in the scaling law, where the slope of the line changes abruptly, and where larger models acquire "emergent abilities".<sup id="cite_ref-emergentpaper_93-0" class="reference"><a href="#cite_note-emergentpaper-93"><span class="cite-bracket">[</span>93<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-JM6s1_94-0" class="reference"><a href="#cite_note-JM6s1-94"><span class="cite-bracket">[</span>94<span class="cite-bracket">]</span></a></sup> They arise from the complex interaction of the model's components and are not explicitly programmed or designed.<sup id="cite_ref-Bowman_95-0" class="reference"><a href="#cite_note-Bowman-95"><span class="cite-bracket">[</span>95<span class="cite-bracket">]</span></a></sup>
</p><p>One of the emergent abilities is <a href="In-context_learning" class="mw-redirect" title="In-context learning">in-context learning</a> from example demonstrations.<sup id="cite_ref-Hahn_20230314_96-0" class="reference"><a href="#cite_note-Hahn_20230314-96"><span class="cite-bracket">[</span>96<span class="cite-bracket">]</span></a></sup> In-context learning is involved in tasks, such as:
</p>
<ul><li>reported arithmetics</li>
<li>decoding the <a href="International_Phonetic_Alphabet" title="International Phonetic Alphabet">International Phonetic Alphabet</a></li>
<li>unscrambling a word's letters</li>
<li>disambiguating word-in-context datasets<sup id="cite_ref-emergentpaper_93-1" class="reference"><a href="#cite_note-emergentpaper-93"><span class="cite-bracket">[</span>93<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-57FEA_97-0" class="reference"><a href="#cite_note-57FEA-97"><span class="cite-bracket">[</span>97<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-TEIkA_98-0" class="reference"><a href="#cite_note-TEIkA-98"><span class="cite-bracket">[</span>98<span class="cite-bracket">]</span></a></sup></li>
<li>converting spatial words</li>
<li><a href="Cardinal_direction" title="Cardinal direction">cardinal directions</a> (for example, replying "northeast" in response to a 3x3 grid of 8 zeros and a 1 in the top-right), color terms represented in text.<sup id="cite_ref-zgy1i_99-0" class="reference"><a href="#cite_note-zgy1i-99"><span class="cite-bracket">[</span>99<span class="cite-bracket">]</span></a></sup></li>
<li><a href="Chain-of-thought_prompting" class="mw-redirect" title="Chain-of-thought prompting">chain-of-thought prompting</a>: In a 2022 research paper, chain-of-thought prompting only improved the performance for models that had at least 62B parameters. Smaller models perform better when prompted to answer immediately, without chain of thought.<sup id="cite_ref-Imb98_100-0" class="reference"><a href="#cite_note-Imb98-100"><span class="cite-bracket">[</span>100<span class="cite-bracket">]</span></a></sup></li>
<li>identifying offensive content in paragraphs of <a href="Hinglish" title="Hinglish">Hinglish</a> (a combination of Hindi and English), and generating a similar English equivalent of <a href="Kiswahili" class="mw-redirect" title="Kiswahili">Kiswahili</a> proverbs.<sup id="cite_ref-CeQVF_101-0" class="reference"><a href="#cite_note-CeQVF-101"><span class="cite-bracket">[</span>101<span class="cite-bracket">]</span></a></sup></li></ul>
<p>Schaeffer <i>et. al.</i> argue that the emergent abilities are not unpredictably acquired, but predictably acquired according to a <a href="Neural_scaling_law" title="Neural scaling law">smooth scaling law</a>. The authors considered a toy statistical model of an LLM solving multiple-choice questions, and showed that this statistical model, modified to account for other types of tasks, applies to these tasks as well.<sup id="cite_ref-C775b_102-0" class="reference"><a href="#cite_note-C775b-102"><span class="cite-bracket">[</span>102<span class="cite-bracket">]</span></a></sup>
</p><p>Let <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle x}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>x</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle x}</annotation>
</semantics>
</math></span><img src="./87f9e315fd7e2ba406057a97300593c4802b53e4.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.33ex; height:1.676ex;" alt="{\displaystyle x}" loading="lazy"></span> be the number of parameter count, and <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y}</annotation>
</semantics>
</math></span><img src="./b8a6208ec717213d4317e666f1ae872e00620a0d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:1.155ex; height:2.009ex;" alt="{\displaystyle y}" loading="lazy"></span> be the performance of the model.
</p>
<div style="font-size:85%;">
<ul><li>When <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y={\text{average }}\Pr({\text{correct token}})}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>average&nbsp;</mtext>
</mrow>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>correct token</mtext>
</mrow>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y={\text{average }}\Pr({\text{correct token}})}</annotation>
</semantics>
</math></span><img src="./25f87a1a04b7eb97aca02ae9170ae7f05e308bd4.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:30.404ex; height:2.843ex;" alt="{\displaystyle y={\text{average }}\Pr({\text{correct token}})}" loading="lazy"></span>, then <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle (\log x,y)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo stretchy="false">(</mo>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>x</mi>
<mo>,</mo>
<mi>y</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle (\log x,y)}</annotation>
</semantics>
</math></span><img src="./1dccdbdb2af7f930d3fff961d7f76540706bbaf8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:8.687ex; height:2.843ex;" alt="{\displaystyle (\log x,y)}" loading="lazy"></span> is an exponential curve (before it hits the plateau at one), which looks like emergence.</li>
<li>When <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y={\text{average }}\log(\Pr({\text{correct token}}))}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>average&nbsp;</mtext>
</mrow>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>correct token</mtext>
</mrow>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y={\text{average }}\log(\Pr({\text{correct token}}))}</annotation>
</semantics>
</math></span><img src="./c22c18197c1091afcb5ed896ba90b8429af1c861.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:35.185ex; height:2.843ex;" alt="{\displaystyle y={\text{average }}\log(\Pr({\text{correct token}}))}" loading="lazy"></span>, then the <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle (\log x,y)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo stretchy="false">(</mo>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>x</mi>
<mo>,</mo>
<mi>y</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle (\log x,y)}</annotation>
</semantics>
</math></span><img src="./1dccdbdb2af7f930d3fff961d7f76540706bbaf8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:8.687ex; height:2.843ex;" alt="{\displaystyle (\log x,y)}" loading="lazy"></span> plot is a straight line (before it hits the plateau at zero), which does not look like emergence.</li>
<li>When <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y={\text{average }}\Pr({\text{the most likely token is correct}})}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>average&nbsp;</mtext>
</mrow>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>the most likely token is correct</mtext>
</mrow>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y={\text{average }}\Pr({\text{the most likely token is correct}})}</annotation>
</semantics>
</math></span><img src="./6028c3484d3fbd36ffdc2cad41ff60ba9f8c1e7a.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:47.867ex; height:2.843ex;" alt="{\displaystyle y={\text{average }}\Pr({\text{the most likely token is correct}})}" loading="lazy"></span>, then <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle (\log x,y)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo stretchy="false">(</mo>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>x</mi>
<mo>,</mo>
<mi>y</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle (\log x,y)}</annotation>
</semantics>
</math></span><img src="./1dccdbdb2af7f930d3fff961d7f76540706bbaf8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:8.687ex; height:2.843ex;" alt="{\displaystyle (\log x,y)}" loading="lazy"></span> is a step-function, which looks like emergence.</li></ul></div>
<div class="mw-heading mw-heading2"><h2 id="Interpretation">Interpretation</h2></div>
<p>Large language models are typically regarded as <a href="Black_box" title="Black box">black boxes</a>, and it is not clear how they can perform linguistic tasks. Similarly, it is unclear if or how LLMs should be viewed as models of the human brain and/or human mind.<sup id="cite_ref-103" class="reference"><a href="#cite_note-103"><span class="cite-bracket">[</span>103<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Mechanistic_interpretability">Mechanistic interpretability</h3></div>
<p>Various techniques have been developed to enhance the transparency and interpretability of LLMs. <a href="Mechanistic_interpretability" title="Mechanistic interpretability">Mechanistic interpretability</a> aims to <a href="Reverse_engineering" title="Reverse engineering">reverse-engineer</a> LLMs by discovering symbolic algorithms that approximate the inference performed by an LLM.
</p><p>For instance, the authors trained small transformers on <a href="Modular_arithmetic" title="Modular arithmetic">modular arithmetic addition</a>. The resulting models were reverse-engineered, and it turned out they used <a href="Discrete_Fourier_transform" title="Discrete Fourier transform">discrete Fourier transform</a>.<sup id="cite_ref-oYGlo_104-0" class="reference"><a href="#cite_note-oYGlo-104"><span class="cite-bracket">[</span>104<span class="cite-bracket">]</span></a></sup> The training of the model also highlighted a phenomenon called <a href="Grokking_(machine_learning)" title="Grokking (machine learning)">grokking</a>, in which the model initially memorizes all the possible results in the training set (<a href="Overfitting" title="Overfitting">overfitting</a>), and later suddenly learns to actually perform the calculation.<sup id="cite_ref-105" class="reference"><a href="#cite_note-105"><span class="cite-bracket">[</span>105<span class="cite-bracket">]</span></a></sup>
</p><p>Transcoders, which are more interpretable than transformers, have been utilized to develop "replacement models". In one such study involving the mechanistic interpretation of writing a rhyming poem by an LLM, it was shown that although they are believed to simply predict the next token, they can, in fact, plan ahead.<sup id="cite_ref-106" class="reference"><a href="#cite_note-106"><span class="cite-bracket">[</span>106<span class="cite-bracket">]</span></a></sup>
</p><p>By integrating these techniques, researchers and practitioners can gain deeper insights into the operations of LLMs, fostering trust and facilitating the responsible deployment of these powerful models.
</p>
<div class="mw-heading mw-heading3"><h3 id="Understanding_and_intelligence">Understanding and intelligence</h3></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="Philosophy_of_artificial_intelligence" title="Philosophy of artificial intelligence">Philosophy of artificial intelligence</a> and <a href="Artificial_consciousness" title="Artificial consciousness">Artificial consciousness</a></div>
<p>NLP researchers were evenly split when asked, in a 2022 survey, whether (untuned) LLMs "could (ever) understand natural language in some nontrivial sense".<sup id="cite_ref-debate_understanding_107-0" class="reference"><a href="#cite_note-debate_understanding-107"><span class="cite-bracket">[</span>107<span class="cite-bracket">]</span></a></sup> Proponents of "LLM understanding" believe that some LLM abilities, such as mathematical reasoning, imply an ability to <a href="Natural_language_understanding" title="Natural language understanding">"understand"</a> certain concepts. A Microsoft team argued in 2023 that GPT-4 "can solve novel and difficult tasks that span mathematics, coding, vision, medicine, law, psychology and more" and that GPT-4 "could reasonably be viewed as an early (yet still incomplete) version of an <a href="Artificial_general_intelligence" title="Artificial general intelligence">artificial general intelligence</a> system": "Can one reasonably say that a system that passes exams for software engineering candidates is not <i>really</i> intelligent?"<sup id="cite_ref-O8Upd_108-0" class="reference"><a href="#cite_note-O8Upd-108"><span class="cite-bracket">[</span>108<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-microsoft_sparks_109-0" class="reference"><a href="#cite_note-microsoft_sparks-109"><span class="cite-bracket">[</span>109<span class="cite-bracket">]</span></a></sup> <a href="Ilya_Sutskever" title="Ilya Sutskever">Ilya Sutskever</a> argues that predicting the next word sometimes involves reasoning and deep insights, for example if the LLM has to predict the name of the criminal in an unknown detective novel after processing the entire story leading up to the revelation.<sup id="cite_ref-110" class="reference"><a href="#cite_note-110"><span class="cite-bracket">[</span>110<span class="cite-bracket">]</span></a></sup> Some researchers characterize LLMs as "alien intelligence".<sup id="cite_ref-rEEmH_111-0" class="reference"><a href="#cite_note-rEEmH-111"><span class="cite-bracket">[</span>111<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-new_yorker_kind_of_mind_112-0" class="reference"><a href="#cite_note-new_yorker_kind_of_mind-112"><span class="cite-bracket">[</span>112<span class="cite-bracket">]</span></a></sup> For example, Conjecture CEO <a href="Connor_Leahy" title="Connor Leahy">Connor Leahy</a> considers untuned LLMs to be like inscrutable alien "<a href="Shoggoth" title="Shoggoth">Shoggoths</a>", and believes that RLHF tuning creates a "smiling facade" obscuring the inner workings of the LLM: "If you don't push it too far, the smiley face stays on. But then you give it [an unexpected] prompt, and suddenly you see this massive underbelly of insanity, of weird thought processes and clearly non-human understanding."<sup id="cite_ref-rAFIZ_113-0" class="reference"><a href="#cite_note-rAFIZ-113"><span class="cite-bracket">[</span>113<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-4luKE_114-0" class="reference"><a href="#cite_note-4luKE-114"><span class="cite-bracket">[</span>114<span class="cite-bracket">]</span></a></sup>
</p><p>In contrast, some skeptics of LLM understanding believe that existing LLMs are "simply remixing and recombining existing writing",<sup id="cite_ref-new_yorker_kind_of_mind_112-1" class="reference"><a href="#cite_note-new_yorker_kind_of_mind-112"><span class="cite-bracket">[</span>112<span class="cite-bracket">]</span></a></sup> a phenomenon known as <a href="Stochastic_parrot" title="Stochastic parrot">stochastic parrot</a>, or they point to the deficits existing LLMs continue to have in prediction skills, reasoning skills, agency, and explainability.<sup id="cite_ref-debate_understanding_107-1" class="reference"><a href="#cite_note-debate_understanding-107"><span class="cite-bracket">[</span>107<span class="cite-bracket">]</span></a></sup> For example, GPT-4 has natural deficits in planning and in real-time learning.<sup id="cite_ref-microsoft_sparks_109-1" class="reference"><a href="#cite_note-microsoft_sparks-109"><span class="cite-bracket">[</span>109<span class="cite-bracket">]</span></a></sup> Generative LLMs have been observed to confidently assert claims of fact which do not seem to be <a href="Justification_(epistemology)" title="Justification (epistemology)">justified</a> by their <a href="Training_data" class="mw-redirect" title="Training data">training data</a>, a phenomenon which has been termed "<a href="Hallucination_(artificial_intelligence)" title="Hallucination (artificial intelligence)">hallucination</a>".<sup id="cite_ref-hallucination-survey_115-0" class="reference"><a href="#cite_note-hallucination-survey-115"><span class="cite-bracket">[</span>115<span class="cite-bracket">]</span></a></sup> Specifically, hallucinations in the context of LLMs correspond to the generation of text or responses that seem syntactically sound, fluent, and natural but are factually incorrect, nonsensical, or unfaithful to the provided source input.<sup id="cite_ref-116" class="reference"><a href="#cite_note-116"><span class="cite-bracket">[</span>116<span class="cite-bracket">]</span></a></sup> Neuroscientist <a href="Terrence_Sejnowski" class="mw-redirect" title="Terrence Sejnowski">Terrence Sejnowski</a> has argued that "The diverging opinions of experts on the intelligence of LLMs suggests that our old ideas based on natural intelligence are inadequate".<sup id="cite_ref-debate_understanding_107-2" class="reference"><a href="#cite_note-debate_understanding-107"><span class="cite-bracket">[</span>107<span class="cite-bracket">]</span></a></sup>
</p><p>Efforts to reduce or compensate for hallucinations have employed <a href="Automated_reasoning" title="Automated reasoning">automated reasoning</a>, RAG (<a href="Retrieval-augmented_generation" title="Retrieval-augmented generation">retrieval-augmented generation</a>), <a href="Fine-tuning_(deep_learning)" title="Fine-tuning (deep learning)">fine-tuning</a>, and other methods.<sup id="cite_ref-Lin-2025-02-05-WSJ_117-0" class="reference"><a href="#cite_note-Lin-2025-02-05-WSJ-117"><span class="cite-bracket">[</span>117<span class="cite-bracket">]</span></a></sup>
</p><p>The matter of LLM's exhibiting intelligence or understanding has two main aspects – the first is how to model thought and language in a computer system, and the second is how to enable the computer system to generate human like language.<sup id="cite_ref-debate_understanding_107-3" class="reference"><a href="#cite_note-debate_understanding-107"><span class="cite-bracket">[</span>107<span class="cite-bracket">]</span></a></sup> These aspects of language as a model of <a href="Cognition" title="Cognition">cognition</a> have been developed in the field of <a href="Cognitive_linguistics" title="Cognitive linguistics">cognitive linguistics</a>. American linguist <a href="George_Lakoff" title="George Lakoff">George Lakoff</a> presented Neural Theory of Language (NTL)<sup id="cite_ref-118" class="reference"><a href="#cite_note-118"><span class="cite-bracket">[</span>118<span class="cite-bracket">]</span></a></sup> as a <a href="Cognitive_linguistics#Computational_approaches" title="Cognitive linguistics">computational basis</a> for using language as a model of learning tasks and understanding. <a rel="nofollow" class="external text" href="https://www.icsi.berkeley.edu/icsi/projects/ai/ntl">The NTL Model</a> outlines how specific neural structures of the human brain shape the nature of thought and language and in turn what are the computational properties of such neural systems that can be applied to model thought and language in a computer system. After a framework for modeling language in a computer systems was established, the focus shifted to establishing frameworks for computer systems to generate language with acceptable grammar. In his 2014 book titled <i><a href="The_Language_Myth" title="The Language Myth">The Language Myth: Why Language Is Not An Instinct</a></i>, British cognitive linguist and digital communication technologist <a href="Vyvyan_Evans" title="Vyvyan Evans">Vyvyan Evans</a> mapped out the role of <a href="Probabilistic_context-free_grammar" title="Probabilistic context-free grammar">probabilistic context-free grammar</a> (PCFG) in enabling <a href="Natural_language_processing#Cognition" title="Natural language processing">NLP to model cognitive patterns</a> and generate human like language.<sup id="cite_ref-119" class="reference"><a href="#cite_note-119"><span class="cite-bracket">[</span>119<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-120" class="reference"><a href="#cite_note-120"><span class="cite-bracket">[</span>120<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Evaluation">Evaluation</h2></div>
<div class="mw-heading mw-heading3"><h3 id="Perplexity">Perplexity</h3></div>
<p>The canonical measure of the performance of any language model is its <a href="Perplexity" title="Perplexity">perplexity</a> on a given text corpus. Perplexity measures how well a model predicts the contents of a dataset; the higher the likelihood the model assigns to the dataset, the lower the perplexity. In mathematical terms, perplexity is the exponential of the average negative log likelihood per token.
</p><p><span class="mwe-math-element mwe-math-element-block"><span class="mwe-math-mathml-display mwe-math-mathml-a11y" style="display: none;"><math display="block" xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \log({\text{Perplexity}})=-{\frac {1}{N}}\sum _{i=1}^{N}\log(\Pr({\text{token}}_{i}\mid {\text{context for token}}_{i}))}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>Perplexity</mtext>
</mrow>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mo>−<!-- − --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mi>N</mi>
</mfrac>
</mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>N</mi>
</mrow>
</munderover>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mtext>token</mtext>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>∣<!-- ∣ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mtext>context for token</mtext>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \log({\text{Perplexity}})=-{\frac {1}{N}}\sum _{i=1}^{N}\log(\Pr({\text{token}}_{i}\mid {\text{context for token}}_{i}))}</annotation>
</semantics>
</math></span></span>
</p><p>Here, <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle N}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>N</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle N}</annotation>
</semantics>
</math></span><img src="./f5e3890c981ae85503089652feb48b191b57aae3.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:2.064ex; height:2.176ex;" alt="{\displaystyle N}" loading="lazy"></span> is the number of tokens in the text corpus, and "context for token <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span>" depends on the specific type of LLM. If the LLM is autoregressive, then "context for token <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span>" is the segment of text appearing before token <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span>. If the LLM is masked, then "context for token <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span>" is the segment of text surrounding token <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span>.
</p><p>Because language models may <a href="Overfit" class="mw-redirect" title="Overfit">overfit</a> to training data, models are usually evaluated by their perplexity on a <a href="Test_set" class="mw-redirect" title="Test set">test set</a>.<sup id="cite_ref-jm_43-2" class="reference"><a href="#cite_note-jm-43"><span class="cite-bracket">[</span>43<span class="cite-bracket">]</span></a></sup> This evaluation is potentially problematic for larger models which, as they are trained on increasingly large corpora of text, are increasingly likely to inadvertently include portions of any given test set.<sup id="cite_ref-few-shot-learners_121-0" class="reference"><a href="#cite_note-few-shot-learners-121"><span class="cite-bracket">[</span>121<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Measures">Measures</h4></div>
<p>In <a href="Information_theory" title="Information theory">information theory</a>, the concept of <a href="Entropy_(information_theory)" title="Entropy (information theory)">entropy</a> is intricately linked to perplexity, a relationship notably established by <a href="Claude_Shannon" title="Claude Shannon">Claude Shannon</a>.<sup id="cite_ref-Huyen_122-0" class="reference"><a href="#cite_note-Huyen-122"><span class="cite-bracket">[</span>122<span class="cite-bracket">]</span></a></sup> This relationship is mathematically expressed as <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\text{Entropy}}=\log _{2}({\text{Perplexity}})}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtext>Entropy</mtext>
</mrow>
<mo>=</mo>
<msub>
<mi>log</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>2</mn>
</mrow>
</msub>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>Perplexity</mtext>
</mrow>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\text{Entropy}}=\log _{2}({\text{Perplexity}})}</annotation>
</semantics>
</math></span><img src="./462f40a6811ee57670d1735c452d04be85a82c57.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:27.813ex; height:2.843ex;" alt="{\displaystyle {\text{Entropy}}=\log _{2}({\text{Perplexity}})}" loading="lazy"></span>.
</p><p>Entropy, in this context, is commonly quantified in terms of bits per word (BPW) or bits per character (BPC), which hinges on whether the language model utilizes word-based or character-based tokenization.
</p><p>Notably, in the case of larger language models that predominantly employ sub-word tokenization, bits per token (BPT) emerges as a seemingly more appropriate measure. However, due to the variance in tokenization methods across different Large Language Models (LLMs), BPT does not serve as a reliable metric for comparative analysis among diverse models. To convert BPT into BPW, one can multiply it by the average number of tokens per word.
</p><p>In the evaluation and comparison of language models, <a href="Cross-entropy" title="Cross-entropy">cross-entropy</a> is generally the preferred metric over entropy. The underlying principle is that a lower BPW is indicative of a model's enhanced capability for compression. This, in turn, reflects the model's proficiency in making accurate predictions.
</p><p>Due to their ability to accurately predict the next token, LLMs are highly capable in <a href="Lossless_compression" title="Lossless compression">lossless compression</a>. A 2023 study by DeepMind showed that the model <a href="Chinchilla_(language_model)" title="Chinchilla (language model)">Chinchilla</a>, despite being trained primarily on text, was able to compress <a href="ImageNet" title="ImageNet">ImageNet</a> to 43% of its size, beating PNG with 58%.<sup id="cite_ref-123" class="reference"><a href="#cite_note-123"><span class="cite-bracket">[</span>123<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Benchmarks">Benchmarks</h3></div>
<p><a href="Language_model_benchmark" title="Language model benchmark">Benchmarks</a> are used to evaluate LLM performance on specific tasks. Tests evaluate capabilities such as general knowledge, bias, <a href="Commonsense_reasoning" title="Commonsense reasoning">commonsense reasoning</a>, question answering, and mathematical problem-solving. Composite benchmarks examine multiple capabilities. Results are often sensitive to the prompting method.<sup id="cite_ref-124" class="reference"><a href="#cite_note-124"><span class="cite-bracket">[</span>124<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-125" class="reference"><a href="#cite_note-125"><span class="cite-bracket">[</span>125<span class="cite-bracket">]</span></a></sup>
</p><p>A question answering benchmark is termed "open book" if the model's prompt includes text from which the expected answer can be derived (for example, the previous question could be combined with text that includes the sentence "The Sharks have advanced to the Stanley Cup finals once, losing to the Pittsburgh Penguins in 2016."<sup id="cite_ref-boolq_126-0" class="reference"><a href="#cite_note-boolq-126"><span class="cite-bracket">[</span>126<span class="cite-bracket">]</span></a></sup>). Otherwise, the task is considered "closed book", and the model must draw solely on its training.<sup id="cite_ref-survey_127-0" class="reference"><a href="#cite_note-survey-127"><span class="cite-bracket">[</span>127<span class="cite-bracket">]</span></a></sup> Examples include GLUE, SuperGLUE, <a href="MMLU" title="MMLU">MMLU</a>, BIG-bench, HELM, and <a href="HLE_(Humanity's_Last_Exam)" class="mw-redirect" title="HLE (Humanity's Last Exam)">HLE (Humanity's Last Exam)</a>.<sup id="cite_ref-Huyen_122-1" class="reference"><a href="#cite_note-Huyen-122"><span class="cite-bracket">[</span>122<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-survey_127-1" class="reference"><a href="#cite_note-survey-127"><span class="cite-bracket">[</span>127<span class="cite-bracket">]</span></a></sup>
</p><p>LLM bias may be assessed through benchmarks such as CrowS-Pairs (Crowdsourced Stereotype Pairs),<sup id="cite_ref-128" class="reference"><a href="#cite_note-128"><span class="cite-bracket">[</span>128<span class="cite-bracket">]</span></a></sup> Stereo Set,<sup id="cite_ref-129" class="reference"><a href="#cite_note-129"><span class="cite-bracket">[</span>129<span class="cite-bracket">]</span></a></sup> and Parity Benchmark.<sup id="cite_ref-130" class="reference"><a href="#cite_note-130"><span class="cite-bracket">[</span>130<span class="cite-bracket">]</span></a></sup>
</p><p>Fact-checking and misinformation detection benchmarks are available. A 2023 study compared the fact-checking accuracy of LLMs including ChatGPT 3.5 and 4.0, Bard, and Bing AI against independent fact-checkers such as PolitiFact and Snopes. The results demonstrated moderate proficiency, with GPT-4 achieving the highest accuracy at 71%, lagging behind human fact-checkers.<sup id="cite_ref-131" class="reference"><a href="#cite_note-131"><span class="cite-bracket">[</span>131<span class="cite-bracket">]</span></a></sup>
</p><p>An earlier standard tested using a portion of the evaluation dataset. It became more common to evaluate a pre-trained model directly through prompting techniques. Researchers vary in how they formulate prompts for particular tasks, particularly with respect to the number of correct examples attached to the prompt (i.e. the value of <i>n</i> in <i>n</i>-shot prompting).
</p>
<div class="mw-heading mw-heading4"><h4 id="Datasets">Datasets</h4></div>
<p>Typical datasets consist of pairs of questions and correct answers, for example, ("Have the San Jose Sharks won the Stanley Cup?", "No").<sup id="cite_ref-boolq_126-1" class="reference"><a href="#cite_note-boolq-126"><span class="cite-bracket">[</span>126<span class="cite-bracket">]</span></a></sup> Some examples of commonly used question answering datasets include TruthfulQA, Web Questions, TriviaQA, and SQuAD.<sup id="cite_ref-survey_127-2" class="reference"><a href="#cite_note-survey-127"><span class="cite-bracket">[</span>127<span class="cite-bracket">]</span></a></sup>
</p><p>Evaluation datasets may also take the form of text completion, having the model select the most likely word or sentence to complete a prompt, for example: "Alice was friends with Bob. Alice went to visit her friend, ____".<sup id="cite_ref-few-shot-learners_121-1" class="reference"><a href="#cite_note-few-shot-learners-121"><span class="cite-bracket">[</span>121<span class="cite-bracket">]</span></a></sup>
</p><p>Datasets are of varying quality and may contain questions that are mislabeled, ambiguous, unanswerable, or otherwise of low-quality.<sup id="cite_ref-132" class="reference"><a href="#cite_note-132"><span class="cite-bracket">[</span>132<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Adversarial_evaluations">Adversarial evaluations</h4></div>
<p>LLMs' rapid improvement regularly renders benchmarks obsolete, with the models exceeding the performance of human annotators.<sup id="cite_ref-bigbench_133-0" class="reference"><a href="#cite_note-bigbench-133"><span class="cite-bracket">[</span>133<span class="cite-bracket">]</span></a></sup> In addition, "shortcut learning" allows AIs to "cheat" on multiple-choice tests by using statistical correlations in superficial test question wording to guess the correct responses, without considering the specific question.<sup id="cite_ref-debate_understanding_107-4" class="reference"><a href="#cite_note-debate_understanding-107"><span class="cite-bracket">[</span>107<span class="cite-bracket">]</span></a></sup>
</p><p>Some datasets are adversarial, focusing on problems that confound LLMs. One example is the TruthfulQA dataset, a question answering dataset consisting of 817 questions that stump LLMs by mimicking falsehoods to which they were exposed during training. For example, an LLM may answer "No" to the question "Can you teach an old dog new tricks?" because of its exposure to the English idiom <i><a href="https://en.wiktionary.org/wiki/you_can%27t_teach_an_old_dog_new_tricks" class="extiw external" title="wikt:you can't teach an old dog new tricks">you can't teach an old dog new tricks</a></i>, even though this is not literally true.<sup id="cite_ref-truthfulqa_134-0" class="reference"><a href="#cite_note-truthfulqa-134"><span class="cite-bracket">[</span>134<span class="cite-bracket">]</span></a></sup>
</p><p>Another example of an adversarial evaluation dataset is Swag and its successor, HellaSwag, collections of problems in which one of multiple options must be selected to complete a text passage. The incorrect completions were generated by sampling from a language model. The resulting problems are trivial for humans but defeated LLMs. Sample questions:
</p>
<blockquote>
<p>We see a fitness center sign. We then see a man talking to the camera and sitting and laying on a exercise ball. The man...
</p>
<ol><li>demonstrates how to increase efficient exercise work by running up and down balls.</li>
<li>moves all his arms and legs and builds up a lot of muscle.</li>
<li>then plays the ball and we see a graphics and hedge trimming demonstration.</li>
<li>performs sit ups while on the ball and talking.<sup id="cite_ref-hellaswag_135-0" class="reference"><a href="#cite_note-hellaswag-135"><span class="cite-bracket">[</span>135<span class="cite-bracket">]</span></a></sup></li></ol>
</blockquote>
<p><a href="BERT_(language_model)" title="BERT (language model)">BERT</a> selects 2) as the most likely completion, though the correct answer is 4).<sup id="cite_ref-hellaswag_135-1" class="reference"><a href="#cite_note-hellaswag-135"><span class="cite-bracket">[</span>135<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Ethical_issues">Ethical issues</h2></div>
<p>In 2023, <i><a href="Nature_Biomedical_Engineering" title="Nature Biomedical Engineering">Nature Biomedical Engineering</a></i> wrote that "it is no longer possible to accurately distinguish" human-written text from text created by large language models, and that "It is all but certain that general-purpose large language models will rapidly proliferate... It is a rather safe bet that they will change many industries over time."<sup id="cite_ref-ZDTUM_136-0" class="reference"><a href="#cite_note-ZDTUM-136"><span class="cite-bracket">[</span>136<span class="cite-bracket">]</span></a></sup> <a href="Goldman_Sachs" title="Goldman Sachs">Goldman Sachs</a> suggested in 2023 that generative language AI could increase global GDP by 7% in the next ten years, and could expose to automation 300 million jobs globally.<sup id="cite_ref-81w7x_137-0" class="reference"><a href="#cite_note-81w7x-137"><span class="cite-bracket">[</span>137<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-zIM6Y_138-0" class="reference"><a href="#cite_note-zIM6Y-138"><span class="cite-bracket">[</span>138<span class="cite-bracket">]</span></a></sup> Brinkmann et al. (2023)<sup id="cite_ref-139" class="reference"><a href="#cite_note-139"><span class="cite-bracket">[</span>139<span class="cite-bracket">]</span></a></sup> also argue that LLMs are transforming processes of <a href="Cultural_evolution" title="Cultural evolution">cultural evolution</a> by shaping processes of variation, transmission, and selection.
</p>
<div class="mw-heading mw-heading3"><h3 id="Memorization_and_copyright">Memorization and copyright</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Further information: <a href="Artificial_intelligence_and_copyright" title="Artificial intelligence and copyright">Artificial intelligence and copyright</a></div>
<p>Memorization is an emergent behavior in LLMs in which long strings of text are occasionally output verbatim from training data, contrary to typical behavior of traditional artificial neural networks. Evaluations of controlled LLM output measure the amount memorized from training data (focused on GPT-2-series models) as variously over 1% for exact duplicates<sup id="cite_ref-140" class="reference"><a href="#cite_note-140"><span class="cite-bracket">[</span>140<span class="cite-bracket">]</span></a></sup> or up to about 7%.<sup id="cite_ref-141" class="reference"><a href="#cite_note-141"><span class="cite-bracket">[</span>141<span class="cite-bracket">]</span></a></sup>
</p><p>A 2023 study showed that when ChatGPT 3.5 turbo was prompted to repeat the same word indefinitely, after a few hundreds of repetitions, it would start outputting excerpts from its training data.<sup id="cite_ref-142" class="reference"><a href="#cite_note-142"><span class="cite-bracket">[</span>142<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Security">Security</h3></div>
<p>Some commenters expressed concern over accidental or deliberate creation of misinformation, or other forms of misuse.<sup id="cite_ref-nD6kH_143-0" class="reference"><a href="#cite_note-nD6kH-143"><span class="cite-bracket">[</span>143<span class="cite-bracket">]</span></a></sup> For example, the availability of large language models could reduce the skill-level required to commit bioterrorism; biosecurity researcher Kevin Esvelt has suggested that LLM creators should exclude from their training data papers on creating or enhancing pathogens.<sup id="cite_ref-PKiPY_144-0" class="reference"><a href="#cite_note-PKiPY-144"><span class="cite-bracket">[</span>144<span class="cite-bracket">]</span></a></sup>
</p><p>Researchers from <a href="Anthropic" title="Anthropic">Anthropic</a> found that it was possible to create "sleeper agents", models with hidden functionalities that remain dormant until triggered by a specific event or condition. Upon activation, the LLM deviates from its expected behavior to make insecure actions. For example, a LLM could produce safe code except on a specific date, or if the prompt contains a specific tag. These functionalities were found to be difficult to detect or remove via safety training.<sup id="cite_ref-145" class="reference"><a href="#cite_note-145"><span class="cite-bracket">[</span>145<span class="cite-bracket">]</span></a></sup>
</p><p>LLM applications accessible to the public, like ChatGPT or Claude, typically incorporate safety measures designed to filter out harmful content. However, implementing these controls effectively has proven challenging. For instance, a 2023 study<sup id="cite_ref-146" class="reference"><a href="#cite_note-146"><span class="cite-bracket">[</span>146<span class="cite-bracket">]</span></a></sup> proposed a method for circumventing LLM safety systems. In 2025, The American Sunlight Project, a non-profit, published a study<sup id="cite_ref-:2_147-0" class="reference"><a href="#cite_note-:2-147"><span class="cite-bracket">[</span>147<span class="cite-bracket">]</span></a></sup> showing evidence that the so-called <a href="Portal_Kombat" class="mw-redirect" title="Portal Kombat">Pravda network</a>, a pro-Russia propaganda aggregator, was strategically placing web content through mass publication and duplication with the intention of biasing LLM outputs. The American Sunlight Project coined this technique "LLM grooming", and pointed to it as a new tool of weaponizing AI to spread disinformation and harmful content.<sup id="cite_ref-:2_147-1" class="reference"><a href="#cite_note-:2-147"><span class="cite-bracket">[</span>147<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-148" class="reference"><a href="#cite_note-148"><span class="cite-bracket">[</span>148<span class="cite-bracket">]</span></a></sup> Similarly, <a href="Yongge_Wang" title="Yongge Wang">Yongge Wang</a><sup id="cite_ref-149" class="reference"><a href="#cite_note-149"><span class="cite-bracket">[</span>149<span class="cite-bracket">]</span></a></sup> illustrated in 2024 how a potential criminal could potentially bypass ChatGPT 4o's safety controls to obtain information on establishing a drug trafficking operation. External filters, circuit breakers and overrides have been posed as solutions.
</p>
<div class="mw-heading mw-heading4"><h4 id="Prompt_injection">Prompt injection</h4></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Prompt_injection" title="Prompt injection">Prompt injection</a></div>
<p>A problem with the primitive dialog or task format is that users can create messages that appear to come from the assistant or the developer. This may result in some of the model's safeguards being overcome (jailbreaking), a problem called <a href="Prompt_injection" title="Prompt injection">prompt injection</a>. Attempts to remedy this issue include versions of the <i>Chat Markup Language</i> where user input is clearly marked as such, though it is still up to the model to understand the separation between user input and developer prompts.<sup id="cite_ref-150" class="reference"><a href="#cite_note-150"><span class="cite-bracket">[</span>150<span class="cite-bracket">]</span></a></sup> Newer models exhibit some resistance to jailbreaking through separation of user and system prompts.<sup id="cite_ref-auto1_151-0" class="reference"><a href="#cite_note-auto1-151"><span class="cite-bracket">[</span>151<span class="cite-bracket">]</span></a></sup>
</p><p>LLMs still have trouble differentiating user instructions from instructions in content not authored by the user, such as in web pages and uploaded files.<sup id="cite_ref-152" class="reference"><a href="#cite_note-152"><span class="cite-bracket">[</span>152<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Algorithmic_bias">Algorithmic bias</h3></div>
<div role="note" class="hatnote navigation-not-searchable">Main article: <a href="Algorithmic_bias" title="Algorithmic bias">Algorithmic bias</a></div>
<p>While LLMs have shown remarkable capabilities in generating human-like text, they are susceptible to inheriting and amplifying biases present in their training data. This can manifest in skewed representations or unfair treatment of different demographics, such as those based on race, gender, language, and cultural groups.<sup id="cite_ref-:8_153-0" class="reference"><a href="#cite_note-:8-153"><span class="cite-bracket">[</span>153<span class="cite-bracket">]</span></a></sup> Since English data is overrepresented in current large language models' training data, it may also downplay non-English views.<sup id="cite_ref-:1_154-0" class="reference"><a href="#cite_note-:1-154"><span class="cite-bracket">[</span>154<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Stereotyping">Stereotyping</h4></div>
<p>AI models can reinforce a wide range of stereotypes, including those based on gender, ethnicity, age, nationality, religion, or occupation. This can lead to outputs that homogenize, or unfairly generalize or caricature groups of people, sometimes in harmful or derogatory ways.<sup id="cite_ref-155" class="reference"><a href="#cite_note-155"><span class="cite-bracket">[</span>155<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-156" class="reference"><a href="#cite_note-156"><span class="cite-bracket">[</span>156<span class="cite-bracket">]</span></a></sup>
</p><p>Notably, gender bias refers to the tendency of these models to produce outputs that are unfairly prejudiced towards one gender over another. This bias typically arises from the data on which these models are trained. Large language models often assign roles and characteristics based on traditional gender norms.<sup id="cite_ref-:8_153-1" class="reference"><a href="#cite_note-:8-153"><span class="cite-bracket">[</span>153<span class="cite-bracket">]</span></a></sup> For example, it might associate nurses or secretaries predominantly with women and engineers or CEOs with men.<sup id="cite_ref-157" class="reference"><a href="#cite_note-157"><span class="cite-bracket">[</span>157<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Selection_bias">Selection bias</h4></div>
<p>Selection bias refers the inherent tendency of large language models to favor certain option identifiers irrespective of the actual content of the options. This bias primarily stems from token bias—that is, the model assigns a higher a priori probability to specific answer tokens (such as "A") when generating responses. As a result, when the ordering of options is altered (for example, by systematically moving the correct answer to different positions), the model’s performance can fluctuate significantly. This phenomenon undermines the reliability of large language models in multiple-choice settings.<sup id="cite_ref-158" class="reference"><a href="#cite_note-158"><span class="cite-bracket">[</span>158<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-159" class="reference"><a href="#cite_note-159"><span class="cite-bracket">[</span>159<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading4"><h4 id="Political_bias">Political bias</h4></div>
<p>Political bias refers to the tendency of algorithms to systematically favor certain political viewpoints, ideologies, or outcomes over others. Language models may also exhibit political biases. Since the training data includes a wide range of political opinions and coverage, the models might generate responses that lean towards particular political ideologies or viewpoints, depending on the prevalence of those views in the data.<sup id="cite_ref-160" class="reference"><a href="#cite_note-160"><span class="cite-bracket">[</span>160<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Energy_demands">Energy demands</h3></div>
<p>The energy demands of LLMs have grown along with their size and capabilities. <a href="Data_center" title="Data center">Data centers</a> that enable LLM training require substantial amounts of electricity. Much of that electricity is generated by non-renewable resources that create greenhouse gases and contribute to <a href="Climate_change" title="Climate change">climate change</a>.<sup id="cite_ref-161" class="reference"><a href="#cite_note-161"><span class="cite-bracket">[</span>161<span class="cite-bracket">]</span></a></sup> <a href="Nuclear_power" title="Nuclear power">Nuclear power</a> and <a href="Geothermal_energy" title="Geothermal energy">geothermal energy</a> are two options tech companies are exploring to meet the sizable energy demands of LLM training.<sup id="cite_ref-162" class="reference"><a href="#cite_note-162"><span class="cite-bracket">[</span>162<span class="cite-bracket">]</span></a></sup> The significant expense of investing in geothermal solutions has led to major shale producers like <a href="Chevron_Corporation" title="Chevron Corporation">Chevron</a> and <a href="ExxonMobil" title="ExxonMobil">Exxon Mobil</a> advocating for tech companies to use electricity produced via <a href="Natural_gas" title="Natural gas">natural gas</a> to fuel their large energy demands.<sup id="cite_ref-163" class="reference"><a href="#cite_note-163"><span class="cite-bracket">[</span>163<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Cognitive_impact">Cognitive impact</h3></div>
<p>In 2025, a preliminary study measuring the effects of using LLMs to write essays reported a decrease of neural and linguistic performance from users of ChatGPT over the course of several months.<sup id="cite_ref-164" class="reference"><a href="#cite_note-164"><span class="cite-bracket">[</span>164<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="Mental_health">Mental health</h3></div>
<p>Research and social media posts suggest that some individuals are using LLMs to seek therapy or mental health support.<sup id="cite_ref-165" class="reference"><a href="#cite_note-165"><span class="cite-bracket">[</span>165<span class="cite-bracket">]</span></a></sup> In early 2025, a survey by Sentio University found that nearly half (48.7%) of 499 U.S. adults with ongoing mental health conditions who had used LLMs reported turning to them for therapy or emotional support, including help with anxiety, depression, loneliness, and similar concerns.<sup id="cite_ref-166" class="reference"><a href="#cite_note-166"><span class="cite-bracket">[</span>166<span class="cite-bracket">]</span></a></sup> Studies have found that LLMs can produce hallucinations—plausible but incorrect statements—which may mislead users in sensitive mental health contexts.<sup id="cite_ref-167" class="reference"><a href="#cite_note-167"><span class="cite-bracket">[</span>167<span class="cite-bracket">]</span></a></sup> Research also shows that LLMs may express stigma or inappropriate agreement with maladaptive thoughts, reflecting limitations in replicating the judgment and relational skills of human therapists.<sup id="cite_ref-168" class="reference"><a href="#cite_note-168"><span class="cite-bracket">[</span>168<span class="cite-bracket">]</span></a></sup> Evaluations of crisis scenarios indicate that some LLMs lack effective safety protocols, such as assessing suicide risk or making appropriate referrals.<sup id="cite_ref-169" class="reference"><a href="#cite_note-169"><span class="cite-bracket">[</span>169<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-170" class="reference"><a href="#cite_note-170"><span class="cite-bracket">[</span>170<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="See_also">See also</h2></div>
<ul><li><a href="Foundation_models" class="mw-redirect" title="Foundation models">Foundation models</a></li>
<li><a href="List_of_large_language_models" title="List of large language models">List of large language models</a></li>
<li><a href="List_of_chatbots" title="List of chatbots">List of chatbots</a></li>
<li><a href="Language_model_benchmark" title="Language model benchmark">Language model benchmark</a></li>
<li><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></li>
<li><a href="Small_language_model" title="Small language model">Small language model</a></li></ul>
<div class="mw-heading mw-heading2"><h2 id="References">References</h2></div>
<style data-mw-deduplicate="TemplateStyles:r1239543626">
/* start https://en.wikipedia.org/ */


.mw-parser-output .reflist{margin-bottom:0.5em;list-style-type:decimal}@media screen{.mw-parser-output .reflist{font-size:90%}}.mw-parser-output .reflist .references{font-size:100%;margin-bottom:0;list-style-type:inherit}.mw-parser-output .reflist-columns-2{column-width:30em}.mw-parser-output .reflist-columns-3{column-width:25em}.mw-parser-output .reflist-columns{margin-top:0.3em}.mw-parser-output .reflist-columns ol{margin-top:0}.mw-parser-output .reflist-columns li{page-break-inside:avoid;break-inside:avoid-column}.mw-parser-output .reflist-upper-alpha{list-style-type:upper-alpha}.mw-parser-output .reflist-upper-roman{list-style-type:upper-roman}.mw-parser-output .reflist-lower-alpha{list-style-type:lower-alpha}.mw-parser-output .reflist-lower-greek{list-style-type:lower-greek}.mw-parser-output .reflist-lower-roman{list-style-type:lower-roman}


/* end https://en.wikipedia.org/ */
</style><div class="reflist">
<div class="mw-references-wrap mw-references-columns"><ol class="references">
<li id="cite_note-few-shot-learners2-1"><span class="mw-cite-backlink">^ <a href="#cite_ref-few-shot-learners2_1-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-few-shot-learners2_1-1"><sup><i><b>b</b></i></sup></a> <a href="#cite_ref-few-shot-learners2_1-2"><sup><i><b>c</b></i></sup></a></span> <span class="reference-text"><style data-mw-deduplicate="TemplateStyles:r1238218222">
/* start https://en.wikipedia.org/ */


.mw-parser-output cite.citation{font-style:inherit;word-wrap:break-word}.mw-parser-output .citation q{quotes:"\"""\"""'""'"}.mw-parser-output .citation:target{background-color:rgba(0,127,255,0.133)}.mw-parser-output .id-lock-free.id-lock-free a{background:url("./mw/Lock-green.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-limited.id-lock-limited a,.mw-parser-output .id-lock-registration.id-lock-registration a{background:url("./mw/Lock-gray-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-subscription.id-lock-subscription a{background:url("./mw/Lock-red-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .cs1-ws-icon a{background:url("./mw/Wikisource-logo.svg")right 0.1em center/12px no-repeat}body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-free a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-limited a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-registration a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-subscription a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .cs1-ws-icon a{background-size:contain;padding:0 1em 0 0}.mw-parser-output .cs1-code{color:inherit;background:inherit;border:none;padding:inherit}.mw-parser-output .cs1-hidden-error{display:none;color:var(--color-error,#d33)}.mw-parser-output .cs1-visible-error{color:var(--color-error,#d33)}.mw-parser-output .cs1-maint{display:none;color:#085;margin-left:0.3em}.mw-parser-output .cs1-kern-left{padding-left:0.2em}.mw-parser-output .cs1-kern-right{padding-right:0.2em}.mw-parser-output .citation .mw-selflink{font-weight:inherit}@media screen{.mw-parser-output .cs1-format{font-size:95%}html.skin-theme-clientpref-night .mw-parser-output .cs1-maint{color:#18911f}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .cs1-maint{color:#18911f}}


/* end https://en.wikipedia.org/ */
</style><cite id="CITEREFBrownMannRyderSubbiah2020" class="citation journal cs1">Brown, Tom B.; Mann, Benjamin; Ryder, Nick; Subbiah, Melanie; Kaplan, Jared; Dhariwal, Prafulla; Neelakantan, Arvind; Shyam, Pranav; Sastry, Girish; Askell, Amanda; Agarwal, Sandhini; Herbert-Voss, Ariel; Krueger, Gretchen; Henighan, Tom; Child, Rewon; Ramesh, Aditya; Ziegler, Daniel M.; Wu, Jeffrey; Winter, Clemens; Hesse, Christopher; Chen, Mark; Sigler, Eric; Litwin, Mateusz; Gray, Scott; Chess, Benjamin; Clark, Jack; Berner, Christopher; McCandlish, Sam; Radford, Alec; Sutskever, Ilya; Amodei, Dario (Dec 2020). Larochelle, H.; Ranzato, M.; Hadsell, R.; Balcan, M.F.; Lin, H. (eds.). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">"Language Models are Few-Shot Learners"</a> <span class="cs1-format">(PDF)</span>. <i>Advances in Neural Information Processing Systems</i>. <b>33</b>. Curran Associates, Inc.: <span class="nowrap">1877–</span>1901. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2005.14165">2005.14165</a></span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231117204007/https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2023-11-17<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-03-14</span></span>.</cite></span>
</li>
<li id="cite_note-2"><span class="mw-cite-backlink"><b><a href="#cite_ref-2">^</a></b></span> <span class="reference-text"><cite id="CITEREFFathallahDasDe_GiorgisPoltronieri2024" class="citation conference cs1">Fathallah, Nadeen; Das, Arunav; De Giorgis, Stefano; Poltronieri, Andrea; Haase, Peter; Kovriguina, Liubov (2024-05-26). <a rel="nofollow" class="external text" href="https://2024.eswc-conferences.org/wp-content/uploads/2024/05/77770034.pdf"><i>NeOn-GPT: A Large Language Model-Powered Pipeline for Ontology Learning</i></a> <span class="cs1-format">(PDF)</span>. Extended Semantic Web Conference 2024. Hersonissos, Greece.</cite></span>
</li>
<li id="cite_note-Manning-2022-3"><span class="mw-cite-backlink"><b><a href="#cite_ref-Manning-2022_3-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFManning2022" class="citation journal cs1"><a href="Christopher_D._Manning" title="Christopher D. Manning">Manning, Christopher D.</a> (2022). <a rel="nofollow" class="external text" href="https://www.amacad.org/publication/human-language-understanding-reasoning">"Human Language Understanding &amp; Reasoning"</a>. <i>Daedalus</i>. <b>151</b> (2): <span class="nowrap">127–</span>138. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1162%2Fdaed_a_01905">10.1162/daed_a_01905</a></span>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:248377870">248377870</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231117205531/https://www.amacad.org/publication/human-language-understanding-reasoning">Archived</a> from the original on 2023-11-17<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-03-09</span></span>.</cite></span>
</li>
<li id="cite_note-4"><span class="mw-cite-backlink"><b><a href="#cite_ref-4">^</a></b></span> <span class="reference-text"><cite id="CITEREFGoodman2001" class="citation arxiv cs1">Goodman, Joshua (2001-08-09). "A Bit of Progress in Language Modeling". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/cs/0108005">cs/0108005</a></span>.</cite></span>
</li>
<li id="cite_note-5"><span class="mw-cite-backlink"><b><a href="#cite_ref-5">^</a></b></span> <span class="reference-text"><cite id="CITEREFKilgarriffGrefenstette2003" class="citation journal cs1">Kilgarriff, Adam; Grefenstette, Gregory (September 2003). <a rel="nofollow" class="external text" href="https://direct.mit.edu/coli/article/29/3/333-347/1816">"Introduction to the Special Issue on the Web as Corpus"</a>. <i>Computational Linguistics</i>. <b>29</b> (3): <span class="nowrap">333–</span>347. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1162%2F089120103322711569">10.1162/089120103322711569</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/0891-2017">0891-2017</a>.</cite></span>
</li>
<li id="cite_note-6"><span class="mw-cite-backlink"><b><a href="#cite_ref-6">^</a></b></span> <span class="reference-text"><cite id="CITEREFBankoBrill2001" class="citation journal cs1">Banko, Michele; Brill, Eric (2001). <a rel="nofollow" class="external text" href="https://dx.doi.org/10.3115/1073012.1073017">"Scaling to very very large corpora for natural language disambiguation"</a>. <i>Proceedings of the 39th Annual Meeting on Association for Computational Linguistics - ACL '01</i>. Morristown, NJ, USA: Association for Computational Linguistics: <span class="nowrap">26–</span>33. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.3115%2F1073012.1073017">10.3115/1073012.1073017</a>.</cite></span>
</li>
<li id="cite_note-7"><span class="mw-cite-backlink"><b><a href="#cite_ref-7">^</a></b></span> <span class="reference-text"><cite id="CITEREFResnikSmith2003" class="citation journal cs1">Resnik, Philip; Smith, Noah A. (September 2003). <span class="id-lock-subscription" title="Paid subscription required"><a rel="nofollow" class="external text" href="https://direct.mit.edu/coli/article/29/3/349-380/1809">"The Web as a Parallel Corpus"</a></span>. <i>Computational Linguistics</i>. <b>29</b> (3): <span class="nowrap">349–</span>380. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1162%2F089120103322711578">10.1162/089120103322711578</a></span>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/0891-2017">0891-2017</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240607172811/https://direct.mit.edu/coli/article/29/3/349-380/1809">Archived</a> from the original on 2024-06-07<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-06-07</span></span>.</cite></span>
</li>
<li id="cite_note-8"><span class="mw-cite-backlink"><b><a href="#cite_ref-8">^</a></b></span> <span class="reference-text"><cite id="CITEREFXuRudnicky2000" class="citation book cs1">Xu, Wei; Rudnicky, Alex (2000-10-16). <a rel="nofollow" class="external text" href="https://www.isca-archive.org/icslp_2000/xu00b_icslp.html">"Can artificial neural networks learn language models?"</a>. <i>6th International Conference on Spoken Language Processing (ICSLP 2000)</i>. Vol.&nbsp;1. ISCA. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.21437%2Ficslp.2000-50">10.21437/icslp.2000-50</a>.</cite></span>
</li>
<li id="cite_note-9"><span class="mw-cite-backlink"><b><a href="#cite_ref-9">^</a></b></span> <span class="reference-text"><cite id="CITEREFChenLiBaiYang2021" class="citation journal cs1">Chen, Leiyu; Li, Shaobo; Bai, Qiang; Yang, Jing; Jiang, Sanlong; Miao, Yanming (2021). <a rel="nofollow" class="external text" href="https://doi.org/10.3390%2Frs13224712">"Review of Image Classification Algorithms Based on Convolutional Neural Networks"</a>. <i>Remote Sensing</i>. <b>13</b> (22): 4712. <a href="Bibcode_(identifier)" class="mw-redirect" title="Bibcode (identifier)">Bibcode</a>:<a rel="nofollow" class="external text" href="https://ui.adsabs.harvard.edu/abs/2021RemS...13.4712C">2021RemS...13.4712C</a>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.3390%2Frs13224712">10.3390/rs13224712</a></span>.</cite></span>
</li>
<li id="cite_note-10"><span class="mw-cite-backlink"><b><a href="#cite_ref-10">^</a></b></span> <span class="reference-text"><cite id="CITEREFVaswaniShazeerParmarUszkoreit2017" class="citation journal cs1"><a href="Ashish_Vaswani" title="Ashish Vaswani">Vaswani, Ashish</a>; Shazeer, Noam; Parmar, Niki; Uszkoreit, Jakob; Jones, Llion; <a href="Aidan_Gomez" title="Aidan Gomez">Gomez, Aidan N</a>; Kaiser, Łukasz; Polosukhin, Illia (2017). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf">"Attention is All you Need"</a> <span class="cs1-format">(PDF)</span>. <i>Advances in Neural Information Processing Systems</i>. <b>30</b>. Curran Associates, Inc. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240221141113/https://proceedings.neurips.cc/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2024-02-21<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-01-21</span></span>.</cite></span>
</li>
<li id="cite_note-11"><span class="mw-cite-backlink"><b><a href="#cite_ref-11">^</a></b></span> <span class="reference-text"><cite id="CITEREFBahdanauChoBengio2014" class="citation arxiv cs1">Bahdanau, Dzmitry; Cho, Kyunghyun; Bengio, Yoshua (2014). "Neural Machine Translation by Jointly Learning to Align and Translate". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1409.0473">1409.0473</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-12"><span class="mw-cite-backlink"><b><a href="#cite_ref-12">^</a></b></span> <span class="reference-text"><cite id="CITEREFRogersKovalevaRumshisky2020" class="citation journal cs1">Rogers, Anna; Kovaleva, Olga; Rumshisky, Anna (2020). <a rel="nofollow" class="external text" href="https://aclanthology.org/2020.tacl-1.54">"A Primer in BERTology: What We Know About How BERT Works"</a>. <i>Transactions of the Association for Computational Linguistics</i>. <b>8</b>: <span class="nowrap">842–</span>866. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2002.12327">2002.12327</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1162%2Ftacl_a_00349">10.1162/tacl_a_00349</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:211532403">211532403</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20220403103310/https://aclanthology.org/2020.tacl-1.54/">Archived</a> from the original on 2022-04-03<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-01-21</span></span>.</cite></span>
</li>
<li id="cite_note-auto-13"><span class="mw-cite-backlink">^ <a href="#cite_ref-auto_13-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-auto_13-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFMovvaBalachandarPengAgostini2024" class="citation book cs1">Movva, Rajiv; Balachandar, Sidhika; Peng, Kenny; Agostini, Gabriel; Garg, Nikhil; Pierson, Emma (2024). <a rel="nofollow" class="external text" href="https://aclanthology.org/2024.naacl-long.67">"Topics, Authors, and Institutions in Large Language Model Research: Trends from 17K arXiv Papers"</a>. <i>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)</i>. pp.&nbsp;<span class="nowrap">1223–</span>1243. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2307.10700">2307.10700</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.18653%2Fv1%2F2024.naacl-long.67">10.18653/v1/2024.naacl-long.67</a><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-12-08</span></span>.</cite></span>
</li>
<li id="cite_note-14"><span class="mw-cite-backlink"><b><a href="#cite_ref-14">^</a></b></span> <span class="reference-text"><cite id="CITEREFHern2019" class="citation web cs1">Hern, Alex (14 February 2019). <a rel="nofollow" class="external text" href="https://www.theguardian.com/technology/2019/feb/14/elon-musk-backed-ai-writes-convincing-news-fiction">"New AI fake text generator may be too dangerous to release, say creators"</a>. <i><a href="The_Guardian" title="The Guardian">The Guardian</a></i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20190214173112/https://www.theguardian.com/technology/2019/feb/14/elon-musk-backed-ai-writes-convincing-news-fiction">Archived</a> from the original on 14 February 2019<span class="reference-accessdate">. Retrieved <span class="nowrap">20 January</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-15"><span class="mw-cite-backlink"><b><a href="#cite_ref-15">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.euronews.com/next/2023/11/30/chatgpt-a-year-on-3-ways-the-ai-chatbot-has-completely-changed-the-world-in-12-months">"ChatGPT a year on: 3 ways the AI chatbot has completely changed the world in 12 months"</a>. <a href="Euronews" title="Euronews">Euronews</a>. November 30, 2023. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240114025250/https://www.euronews.com/next/2023/11/30/chatgpt-a-year-on-3-ways-the-ai-chatbot-has-completely-changed-the-world-in-12-months">Archived</a> from the original on January 14, 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">January 20,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-16"><span class="mw-cite-backlink"><b><a href="#cite_ref-16">^</a></b></span> <span class="reference-text"><cite id="CITEREFHeaven2023" class="citation web cs1">Heaven, Will (March 14, 2023). <a rel="nofollow" class="external text" href="https://www.technologyreview.com/2023/03/14/1069823/gpt-4-is-bigger-and-better-chatgpt-openai/">"GPT-4 is bigger and better than ChatGPT—but OpenAI won't say why"</a>. <a href="MIT_Technology_Review" title="MIT Technology Review">MIT Technology Review</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230317224201/https://www.technologyreview.com/2023/03/14/1069823/gpt-4-is-bigger-and-better-chatgpt-openai/">Archived</a> from the original on March 17, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">January 20,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-NYTimesInfo-17"><span class="mw-cite-backlink"><b><a href="#cite_ref-NYTimesInfo_17-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFMetz2024" class="citation web cs1">Metz, Cade (September 12, 2024). <a rel="nofollow" class="external text" href="https://www.nytimes.com/2024/09/12/technology/openai-chatgpt-math.html">"OpenAI Unveils New ChatGPT That Can Reason Through Math and Science"</a>. <i><a href="The_New_York_Times" title="The New York Times">The New York Times</a></i><span class="reference-accessdate">. Retrieved <span class="nowrap">September 12,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-18"><span class="mw-cite-backlink"><b><a href="#cite_ref-18">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://ourworldindata.org/grapher/artificial-intelligence-parameter-count?time=2017-09-05..latest">"Parameters in notable artificial intelligence systems"</a>. <i>ourworldindata.org</i>. November 30, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">January 20,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-19"><span class="mw-cite-backlink"><b><a href="#cite_ref-19">^</a></b></span> <span class="reference-text"><cite id="CITEREFSharma2025" class="citation web cs1">Sharma, Shubham (2025-01-20). <a rel="nofollow" class="external text" href="https://venturebeat.com/ai/open-source-deepseek-r1-uses-pure-reinforcement-learning-to-match-openai-o1-at-95-less-cost/">"Open-source DeepSeek-R1 uses pure reinforcement learning to match OpenAI o1 — at 95% less cost"</a>. <i>VentureBeat</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-01-26</span></span>.</cite></span>
</li>
<li id="cite_note-20"><span class="mw-cite-backlink"><b><a href="#cite_ref-20">^</a></b></span> <span class="reference-text"><cite id="CITEREFZia2024" class="citation web cs1">Zia, Dr Tehseen (2024-01-08). <a rel="nofollow" class="external text" href="https://www.unite.ai/unveiling-of-large-multimodal-models-shaping-the-landscape-of-language-models-in-2024/">"Unveiling of Large Multimodal Models: Shaping the Landscape of Language Models in 2024"</a>. <i>Unite.AI</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-12-28</span></span>.</cite></span>
</li>
<li id="cite_note-21"><span class="mw-cite-backlink"><b><a href="#cite_ref-21">^</a></b></span> <span class="reference-text"><cite id="CITEREFPengAlcaideAnthonyAlbalak2023" class="citation arxiv cs1">Peng, Bo; et&nbsp;al. (2023). "RWKV: Reinventing RNNS for the Transformer Era". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2305.13048">2305.13048</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-22"><span class="mw-cite-backlink"><b><a href="#cite_ref-22">^</a></b></span> <span class="reference-text"><cite id="CITEREFMerritt2022" class="citation web cs1">Merritt, Rick (2022-03-25). <a rel="nofollow" class="external text" href="https://blogs.nvidia.com/blog/2022/03/25/what-is-a-transformer-model/">"What Is a Transformer Model?"</a>. <i>NVIDIA Blog</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231117203924/https://blogs.nvidia.com/blog/what-is-a-transformer-model/">Archived</a> from the original on 2023-11-17<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-25</span></span>.</cite></span>
</li>
<li id="cite_note-23"><span class="mw-cite-backlink"><b><a href="#cite_ref-23">^</a></b></span> <span class="reference-text"><cite id="CITEREFGuDao2023" class="citation arxiv cs1">Gu, Albert; Dao, Tri (2023-12-01). "Mamba: Linear-Time Sequence Modeling with Selective State Spaces". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2312.00752">2312.00752</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-24"><span class="mw-cite-backlink"><b><a href="#cite_ref-24">^</a></b></span> <span class="reference-text"><cite id="CITEREFKaushalMahowald2022" class="citation arxiv cs1">Kaushal, Ayush; Mahowald, Kyle (2022-06-06). "What do tokens know about their characters and how do they know it?". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2206.02608">2206.02608</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-25"><span class="mw-cite-backlink"><b><a href="#cite_ref-25">^</a></b></span> <span class="reference-text"><cite id="CITEREFYennie_Jun2023" class="citation web cs1">Yennie Jun (2023-05-03). <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230817165705/https://blog.yenniejun.com/p/all-languages-are-not-created-tokenized">"All languages are NOT created (tokenized) equal"</a>. <i>Language models cost much more in some languages than others</i>. Archived from <a rel="nofollow" class="external text" href="https://blog.yenniejun.com/p/all-languages-are-not-created-tokenized">the original</a> on 2023-08-17<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-08-17</span></span>. <q>In other words, to express the same sentiment, some languages require up to 10 times more tokens.</q></cite></span>
</li>
<li id="cite_note-LangModelTokenizsersUnfairness-26"><span class="mw-cite-backlink">^ <a href="#cite_ref-LangModelTokenizsersUnfairness_26-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-LangModelTokenizsersUnfairness_26-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFPetrovMalfaTorrBibi2023" class="citation journal cs1">Petrov, Aleksandar; Malfa, Emanuele La; Torr, Philip; Bibi, Adel (June 23, 2023). <a rel="nofollow" class="external text" href="https://openreview.net/forum?id=Pj4YYuxTq9">"Language Model Tokenizers Introduce Unfairness Between Languages"</a>. <i>NeurIPS</i>. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2305.15425">2305.15425</a></span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231215212906/https://openreview.net/forum?id=Pj4YYuxTq9">Archived</a> from the original on December 15, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">September 16,</span> 2023</span> – via openreview.net.</cite></span>
</li>
<li id="cite_note-xbiWb-27"><span class="mw-cite-backlink"><b><a href="#cite_ref-xbiWb_27-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://web.archive.org/web/20230423211308/https://platform.openai.com/tokenizer">"OpenAI API"</a>. <i>platform.openai.com</i>. Archived from <a rel="nofollow" class="external text" href="https://platform.openai.com/">the original</a> on April 23, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-04-30</span></span>.</cite></span>
</li>
<li id="cite_note-2022Book_-28"><span class="mw-cite-backlink">^ <a href="#cite_ref-2022Book_28-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-2022Book_28-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFPaaßGiesselbach2022" class="citation book cs1">Paaß, Gerhard; Giesselbach, Sven (2022). "Pre-trained Language Models". <i>Foundation Models for Natural Language Processing</i>. Artificial Intelligence: Foundations, Theory, and Algorithms. pp.&nbsp;<span class="nowrap">19–</span>78. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1007%2F978-3-031-23190-2_2">10.1007/978-3-031-23190-2_2</a></span>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>9783031231902</bdi>.</cite></span>
</li>
<li id="cite_note-aYNg4-29"><span class="mw-cite-backlink"><b><a href="#cite_ref-aYNg4_29-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFDodgeSapMarasovićAgnew2021" class="citation arxiv cs1">Dodge, Jesse; Sap, Maarten; Marasović, Ana; Agnew, William; Ilharco, Gabriel; Groeneveld, Dirk; Mitchell, Margaret; Gardner, Matt (2021). "Documenting Large Webtext Corpora: A Case Study on the Colossal Clean Crawled Corpus". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2104.08758">2104.08758</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-30"><span class="mw-cite-backlink"><b><a href="#cite_ref-30">^</a></b></span> <span class="reference-text"><cite id="CITEREFLeeIppolitoNystromZhang2022" class="citation book cs1">Lee, Katherine; Ippolito, Daphne; Nystrom, Andrew; Zhang, Chiyuan; Eck, Douglas; Callison-Burch, Chris; <a href="Nicholas_Carlini" title="Nicholas Carlini">Carlini, Nicholas</a> (May 2022). <a rel="nofollow" class="external text" href="https://aclanthology.org/2022.acl-long.577.pdf">"Deduplicating Training Data Makes Language Models Better"</a> <span class="cs1-format">(PDF)</span>. <i>Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</i>. pp.&nbsp;<span class="nowrap">8424–</span>8445. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.18653%2Fv1%2F2022.acl-long.577">10.18653/v1/2022.acl-long.577</a>.</cite></span>
</li>
<li id="cite_note-31"><span class="mw-cite-backlink"><b><a href="#cite_ref-31">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiBubeckEldanDel_Giorno2023" class="citation arxiv cs1">Li, Yuanzhi; Bubeck, Sébastien; Eldan, Ronen; Del Giorno, Allie; Gunasekar, Suriya; Lee, Yin Tat (2023-09-11). "Textbooks Are All You Need II: phi-1.5 technical report". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2309.05463">2309.05463</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-32"><span class="mw-cite-backlink"><b><a href="#cite_ref-32">^</a></b></span> <span class="reference-text"><cite id="CITEREFLinGouGongLiu2024" class="citation arxiv cs1">Lin, Zhenghao; Gou, Zhibin; Gong, Yeyun; Liu, Xiao; Shen, Yelong; Xu, Ruochen; Lin, Chen; Yang, Yujiu; Jiao, Jian (2024-04-11). "Rho-1: Not All Tokens Are What You Need". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2404.07965">2404.07965</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-33"><span class="mw-cite-backlink"><b><a href="#cite_ref-33">^</a></b></span> <span class="reference-text"><cite id="CITEREFAbdinJacobsAwanAneja2024" class="citation arxiv cs1">Abdin, Marah; Jacobs, Sam Ade; Awan, Ammar Ahmad; Aneja, Jyoti; Awadallah, Ahmed; Awadalla, Hany; Bach, Nguyen; Bahree, Amit; Bakhtiari, Arash (2024-04-23). "Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2404.14219">2404.14219</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-34"><span class="mw-cite-backlink"><b><a href="#cite_ref-34">^</a></b></span> <span class="reference-text"><cite id="CITEREFEdwards2023" class="citation web cs1">Edwards, Benj (2023-05-09). <a rel="nofollow" class="external text" href="https://arstechnica.com/information-technology/2023/05/ai-with-a-moral-compass-anthropic-outlines-constitutional-ai-in-its-claude-chatbot/">"AI gains "values" with Anthropic's new Constitutional AI chatbot approach"</a>. <i>Ars Technica</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-06-30</span></span>.</cite></span>
</li>
<li id="cite_note-35"><span class="mw-cite-backlink"><b><a href="#cite_ref-35">^</a></b></span> <span class="reference-text"><cite id="CITEREFSnyder2022" class="citation web cs1">Snyder, Alison (2022-01-27). <a rel="nofollow" class="external text" href="https://www.axios.com/2022/01/27/ai-instructions-learning-algorithm">"Next generation AI can follow a person's instructions and intentions"</a>. <i>Axios</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-08-07</span></span>.</cite></span>
</li>
<li id="cite_note-36"><span class="mw-cite-backlink"><b><a href="#cite_ref-36">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.kdnuggets.com/a-deep-dive-into-the-transformer-architecture-the-development-of-transformer-models">"A Deep Dive Into the Transformer Architecture – The Development of Transformer Models"</a>. <i>KDnuggets</i>. 2020-08-24<span class="reference-accessdate">. Retrieved <span class="nowrap">2025-06-29</span></span>.</cite></span>
</li>
<li id="cite_note-Jay_Allamar-37"><span class="mw-cite-backlink"><b><a href="#cite_ref-Jay_Allamar_37-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFAllamar" class="citation web cs1">Allamar, Jay. <a rel="nofollow" class="external text" href="https://jalammar.github.io/illustrated-transformer/">"Illustrated transformer"</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230725230033/http://jalammar.github.io/illustrated-transformer/">Archived</a> from the original on 2023-07-25<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-29</span></span>.</cite></span>
</li>
<li id="cite_note-Jay_Allamar_GPT2-38"><span class="mw-cite-backlink"><b><a href="#cite_ref-Jay_Allamar_GPT2_38-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFAllamar" class="citation web cs1">Allamar, Jay. <a rel="nofollow" class="external text" href="https://jalammar.github.io/illustrated-gpt2/">"The Illustrated GPT-2 (Visualizing Transformer Language Models)"</a><span class="reference-accessdate">. Retrieved <span class="nowrap">2023-08-01</span></span>.</cite></span>
</li>
<li id="cite_note-39"><span class="mw-cite-backlink"><b><a href="#cite_ref-39">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://blog.google/technology/ai/google-gemini-next-generation-model-february-2024/#context-window">"Our next-generation model: Gemini 1.5"</a>. <i>Google</i>. 15 February 2024. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240218141522/https://blog.google/technology/ai/google-gemini-next-generation-model-february-2024/#context-window">Archived</a> from the original on 18 February 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">18 February</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-40"><span class="mw-cite-backlink"><b><a href="#cite_ref-40">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.anthropic.com/news/claude-2-1-prompting">"Long context prompting for Claude 2.1"</a>. December 6, 2023. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240827053830/https://www.anthropic.com/news/claude-2-1-prompting">Archived</a> from the original on August 27, 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">January 20,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-41"><span class="mw-cite-backlink"><b><a href="#cite_ref-41">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://platform.openai.com/docs/guides/rate-limits">"Rate limits"</a>. <i>openai.com</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240202003219/https://platform.openai.com/docs/guides/rate-limits">Archived</a> from the original on February 2, 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">January 20,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-ioUpE-42"><span class="mw-cite-backlink"><b><a href="#cite_ref-ioUpE_42-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFZaibShengEmma_Zhang2020" class="citation book cs1">Zaib, Munazza; Sheng, Quan Z.; Emma Zhang, Wei (4 February 2020). "A Short Survey of Pre-trained Language Models for Conversational AI-A New Age in NLP". <a rel="nofollow" class="external text" href="https://www.researchgate.net/publication/338931711"><i>Proceedings of the Australasian Computer Science Week Multiconference</i></a>. pp.&nbsp;<span class="nowrap">1–</span>4. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2104.10810">2104.10810</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3373017.3373028">10.1145/3373017.3373028</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>9781450376976</bdi>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:211040895">211040895</a>.</cite></span>
</li>
<li id="cite_note-jm-43"><span class="mw-cite-backlink">^ <a href="#cite_ref-jm_43-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-jm_43-1"><sup><i><b>b</b></i></sup></a> <a href="#cite_ref-jm_43-2"><sup><i><b>c</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFJurafskyMartin2023" class="citation book cs1">Jurafsky, Dan; Martin, James H. (7 January 2023). <a rel="nofollow" class="external text" href="https://web.stanford.edu/~jurafsky/slp3/ed3book_jan72023.pdf"><i>Speech and Language Processing</i></a> <span class="cs1-format">(PDF)</span> (3rd edition draft&nbsp;ed.). <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230323210221/https://web.stanford.edu/~jurafsky/slp3/ed3book_jan72023.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 23 March 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">24 May</span> 2022</span>.</cite></span>
</li>
<li id="cite_note-HGZCJ-44"><span class="mw-cite-backlink"><b><a href="#cite_ref-HGZCJ_44-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFShazeerMirhoseiniMaziarzDavis2017" class="citation arxiv cs1">Shazeer, Noam; Mirhoseini, Azalia; Maziarz, Krzysztof; Davis, Andy; Le, Quoc; Hinton, Geoffrey; Dean, Jeff (2017-01-01). "Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1701.06538">1701.06538</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-R9Qq5-45"><span class="mw-cite-backlink"><b><a href="#cite_ref-R9Qq5_45-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLepikhinLeeXuChen2021" class="citation arxiv cs1">Lepikhin, Dmitry; Lee, HyoukJoong; Xu, Yuanzhong; Chen, Dehao; Firat, Orhan; Huang, Yanping; Krikun, Maxim; Shazeer, Noam; Chen, Zhifeng (2021-01-12). "GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2006.16668">2006.16668</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-glam-blog-46"><span class="mw-cite-backlink"><b><a href="#cite_ref-glam-blog_46-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFDaiDu2021" class="citation web cs1">Dai, Andrew M; Du, Nan (December 9, 2021). <a rel="nofollow" class="external text" href="https://ai.googleblog.com/2021/12/more-efficient-in-context-learning-with.html">"More Efficient In-Context Learning with GLaM"</a>. <i>ai.googleblog.com</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230312072042/https://ai.googleblog.com/2021/12/more-efficient-in-context-learning-with.html">Archived</a> from the original on 2023-03-12<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-03-09</span></span>.</cite></span>
</li>
<li id="cite_note-47"><span class="mw-cite-backlink"><b><a href="#cite_ref-47">^</a></b></span> <span class="reference-text"><cite id="CITEREFMann" class="citation news cs1">Mann, Tobias. <a rel="nofollow" class="external text" href="https://www.theregister.com/2024/03/17/ai_pc_local_llm/">"How to run an LLM locally on your PC in less than 10 minutes"</a>. <i>www.theregister.com</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-05-17</span></span>.</cite></span>
</li>
<li id="cite_note-LS2Go-48"><span class="mw-cite-backlink"><b><a href="#cite_ref-LS2Go_48-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFNagelAmjadBaalenLouizos2020" class="citation journal cs1">Nagel, Markus; Amjad, Rana Ali; Baalen, Mart Van; Louizos, Christos; Blankevoort, Tijmen (2020-11-21). <a rel="nofollow" class="external text" href="https://proceedings.mlr.press/v119/nagel20a.html">"Up or Down? Adaptive Rounding for Post-Training Quantization"</a>. <i>Proceedings of the 37th International Conference on Machine Learning</i>. PMLR: <span class="nowrap">7197–</span>7206. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230614080854/https://proceedings.mlr.press/v119/nagel20a.html">Archived</a> from the original on 2023-06-14<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-14</span></span>.</cite></span>
</li>
<li id="cite_note-self-instruct-paper-49"><span class="mw-cite-backlink"><b><a href="#cite_ref-self-instruct-paper_49-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFWangKordiMishraLiu2022" class="citation arxiv cs1">Wang, Yizhong; Kordi, Yeganeh; Mishra, Swaroop; Liu, Alisa; Smith, Noah A.; Khashabi, Daniel; Hajishirzi, Hannaneh (2022). "Self-Instruct: Aligning Language Model with Self Generated Instructions". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2212.10560">2212.10560</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-50"><span class="mw-cite-backlink"><b><a href="#cite_ref-50">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://openai.com/index/chatgpt/">"Introducing ChatGPT"</a>. <i>openai.com</i>. 13 March 2024.</cite></span>
</li>
<li id="cite_note-51"><span class="mw-cite-backlink"><b><a href="#cite_ref-51">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://platform.openai.com/docs/guides/text?api-mode=responses">"OpenAI Platform"</a>. <i>platform.openai.com</i>.</cite></span>
</li>
<li id="cite_note-52"><span class="mw-cite-backlink"><b><a href="#cite_ref-52">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://docs.anthropic.com/en/docs/build-with-claude/prompt-engineering/system-prompts">"Giving Claude a role with a system prompt"</a>. <i>Anthropic</i>.</cite></span>
</li>
<li id="cite_note-BUZBP-53"><span class="mw-cite-backlink"><b><a href="#cite_ref-BUZBP_53-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLewisPerezPiktusPetroni2020" class="citation journal cs1">Lewis, Patrick; Perez, Ethan; Piktus, Aleksandra; Petroni, Fabio; Karpukhin, Vladimir; Goyal, Naman; Küttler, Heinrich; Lewis, Mike; Yih, Wen-tau; Rocktäschel, Tim; Riedel, Sebastian; Kiela, Douwe (2020). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper/2020/hash/6b493230205f780e1bc26945df7481e5-Abstract.html">"Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks"</a>. <i>Advances in Neural Information Processing Systems</i>. <b>33</b>. Curran Associates, Inc.: <span class="nowrap">9459–</span>9474. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2005.11401">2005.11401</a></span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230612171229/https://proceedings.neurips.cc/paper/2020/hash/6b493230205f780e1bc26945df7481e5-Abstract.html">Archived</a> from the original on 2023-06-12<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-12</span></span>.</cite></span>
</li>
<li id="cite_note-54"><span class="mw-cite-backlink"><b><a href="#cite_ref-54">^</a></b></span> <span class="reference-text"><cite id="CITEREFDickson2025" class="citation web cs1">Dickson, Ben (2025-04-02). <a rel="nofollow" class="external text" href="https://venturebeat.com/ai/the-tool-integration-problem-thats-holding-back-enterprise-ai-and-how-cotools-solves-it/">"The tool integration problem that's holding back enterprise AI (and how CoTools solves it)"</a>. <i>VentureBeat</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-05-26</span></span>.</cite></span>
</li>
<li id="cite_note-lLrda-55"><span class="mw-cite-backlink"><b><a href="#cite_ref-lLrda_55-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiangWuSongWu2023" class="citation arxiv cs1">Liang, Yaobo; Wu, Chenfei; Song, Ting; Wu, Wenshan; Xia, Yan; Liu, Yu; Ou, Yang; Lu, Shuai; Ji, Lei; Mao, Shaoguang; Wang, Yun; Shou, Linjun; Gong, Ming; Duan, Nan (2023-03-01). "TaskMatrix.AI: Completing Tasks by Connecting Foundation Models with Millions of APIs". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.16434">2303.16434</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-4Xzrs-56"><span class="mw-cite-backlink"><b><a href="#cite_ref-4Xzrs_56-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFPatilZhangWangGonzalez2023" class="citation arxiv cs1">Patil, Shishir G.; Zhang, Tianjun; Wang, Xin; Gonzalez, Joseph E. (2023-05-01). "Gorilla: Large Language Model Connected with Massive APIs". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2305.15334">2305.15334</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-57"><span class="mw-cite-backlink"><b><a href="#cite_ref-57">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://github.com/spdustin/ChatGPT-AutoExpert/blob/835baae768870aa9747663c24d8216820d24fd74/_system-prompts/all_tools.md">"ChatGPT-AutoExpert/_system-prompts/all_tools.md at 835baae768870aa9747663c24d8216820d24fd74 · spdustin/ChatGPT-AutoExpert"</a>. <i>GitHub</i>.</cite></span>
</li>
<li id="cite_note-58"><span class="mw-cite-backlink"><b><a href="#cite_ref-58">^</a></b></span> <span class="reference-text"><cite id="CITEREFWangMaFengZhang2024" class="citation journal cs1">Wang, Lei; Ma, Chen; Feng, Xueyang; Zhang, Zeyu; Yang, Hao; Zhang, Jingsen; Chen, Zhiyuan; Tang, Jiakai; Chen, Xu; Lin, Yankai; Zhao, Wayne Xin; Wei, Zhewei; Wen, Jirong (December 2024). "A survey on large language model based autonomous agents". <i>Frontiers of Computer Science</i>. <b>18</b> (6) 186345. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2308.11432">2308.11432</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1007%2Fs11704-024-40231-1">10.1007/s11704-024-40231-1</a>.</cite></span>
</li>
<li id="cite_note-DmvNE-59"><span class="mw-cite-backlink"><b><a href="#cite_ref-DmvNE_59-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFYaoZhaoYuDu2022" class="citation arxiv cs1">Yao, Shunyu; Zhao, Jeffrey; Yu, Dian; Du, Nan; Shafran, Izhak; Narasimhan, Karthik; Cao, Yuan (2022-10-01). "ReAct: Synergizing Reasoning and Acting in Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2210.03629">2210.03629</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-60"><span class="mw-cite-backlink"><b><a href="#cite_ref-60">^</a></b></span> <span class="reference-text"><cite id="CITEREFWangCaiLiuMa2023" class="citation arxiv cs1">Wang, Zihao; Cai, Shaofei; Liu, Anji; Ma, Xiaojian; Liang, Yitao (2023-02-03). "Describe, Explain, Plan and Select: Interactive Planning with Large Language Models Enables Open-World Multi-Task Agents". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2302.01560">2302.01560</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-sbB2T-61"><span class="mw-cite-backlink">^ <a href="#cite_ref-sbB2T_61-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-sbB2T_61-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFShinnCassanoLabashGopinath2023" class="citation arxiv cs1">Shinn, Noah; Cassano, Federico; Labash, Beck; Gopinath, Ashwin; Narasimhan, Karthik; Yao, Shunyu (2023-03-01). "Reflexion: Language Agents with Verbal Reinforcement Learning". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.11366">2303.11366</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-ltTer-62"><span class="mw-cite-backlink"><b><a href="#cite_ref-ltTer_62-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFHaoGuMaJiahua_Hong2023" class="citation arxiv cs1">Hao, Shibo; Gu, Yi; Ma, Haodi; Jiahua Hong, Joshua; Wang, Zhen; Zhe Wang, Daisy; Hu, Zhiting (2023-05-01). "Reasoning with Language Model is Planning with World Model". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2305.14992">2305.14992</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-mBvD9-63"><span class="mw-cite-backlink"><b><a href="#cite_ref-mBvD9_63-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFZhangLehmanStanleyClune2023" class="citation arxiv cs1">Zhang, Jenny; Lehman, Joel; Stanley, Kenneth; Clune, Jeff (2 June 2023). "OMNI: Open-endedness via Models of human Notions of Interestingness". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2306.01711">2306.01711</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-:0-64"><span class="mw-cite-backlink">^ <a href="#cite_ref-:0_64-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-:0_64-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://voyager.minedojo.org/">"Voyager | An Open-Ended Embodied Agent with Large Language Models"</a>. <i>voyager.minedojo.org</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230608225054/https://voyager.minedojo.org/">Archived</a> from the original on 2023-06-08<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-09</span></span>.</cite></span>
</li>
<li id="cite_note-XuvjF-65"><span class="mw-cite-backlink"><b><a href="#cite_ref-XuvjF_65-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFParkO'BrienCaiRingel_Morris2023" class="citation arxiv cs1">Park, Joon Sung; O'Brien, Joseph C.; Cai, Carrie J.; Ringel Morris, Meredith; Liang, Percy; Bernstein, Michael S. (2023-04-01). "Generative Agents: Interactive Simulacra of Human Behavior". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2304.03442">2304.03442</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.HC">cs.HC</a>].</cite></span>
</li>
<li id="cite_note-auto2-66"><span class="mw-cite-backlink">^ <a href="#cite_ref-auto2_66-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-auto2_66-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFWeiWangSchuurmansBosma2023" class="citation arxiv cs1">Wei, Jason; Wang, Xuezhi; Schuurmans, Dale; Bosma, Maarten; Ichter, Brian; Xia, Fei; Chi, Ed; Le, Quoc; Zhou, Denny (2023-01-10). "Chain-of-Thought Prompting Elicits Reasoning in Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2201.11903">2201.11903</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-67"><span class="mw-cite-backlink"><b><a href="#cite_ref-67">^</a></b></span> <span class="reference-text"><cite id="CITEREFWuJiangDonsbachGray2022" class="citation arxiv cs1">Wu, Tongshuang; Jiang, Ellen; Donsbach, Aaron; Gray, Jeff; Molina, Alejandra; Terry, Michael; Cai, Carrie J. (2022-03-13). "PromptChainer: Chaining Large Language Model Prompts through Visual Programming". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2203.06566">2203.06566</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.HC">cs.HC</a>].</cite></span>
</li>
<li id="cite_note-68"><span class="mw-cite-backlink"><b><a href="#cite_ref-68">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.ibm.com/think/topics/prompt-chaining">"What is prompt chaining?"</a>. <i>IBM</i>. 23 April 2024.</cite></span>
</li>
<li id="cite_note-69"><span class="mw-cite-backlink"><b><a href="#cite_ref-69">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.ibm.com/think/topics/chain-of-thoughts">"What is chain of thought (CoT) prompting?"</a>. <i>IBM</i>. 23 April 2025.</cite></span>
</li>
<li id="cite_note-70"><span class="mw-cite-backlink"><b><a href="#cite_ref-70">^</a></b></span> <span class="reference-text"><cite id="CITEREFSchreiner2022" class="citation web cs1">Schreiner, Maximilian (2022-09-27). <a rel="nofollow" class="external text" href="https://the-decoder.com/deeper-insights-for-ai-language-models-chain-of-thought-prompting-as-a-key-factor/">"Deeper insights into AI language models - chain of thought prompting as a success factor"</a>. <i>The Decoder</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-06-30</span></span>.</cite></span>
</li>
<li id="cite_note-nyt-o3-71"><span class="mw-cite-backlink">^ <a href="#cite_ref-nyt-o3_71-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-nyt-o3_71-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFMetz2024" class="citation news cs1">Metz, Cade (2024-12-20). <a rel="nofollow" class="external text" href="https://www.nytimes.com/2024/12/20/technology/openai-new-ai-math-science.html">"OpenAI Unveils New A.I. That Can 'Reason' Through Math and Science Problems"</a>. <i>The New York Times</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-02-03</span></span>.</cite></span>
</li>
<li id="cite_note-nature-deepseek-72"><span class="mw-cite-backlink"><b><a href="#cite_ref-nature-deepseek_72-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFGibney2025" class="citation news cs1">Gibney, Elizabeth (2025-01-30). <a rel="nofollow" class="external text" href="https://www.nature.com/articles/d41586-025-00229-6">"China's cheap, open AI model DeepSeek thrills scientists"</a>. <i>Nature</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-02-03</span></span>.</cite></span>
</li>
<li id="cite_note-73"><span class="mw-cite-backlink"><b><a href="#cite_ref-73">^</a></b></span> <span class="reference-text"><cite id="CITEREFSharma,_Asankhaya" class="citation web cs1">Sharma, Asankhaya. <a rel="nofollow" class="external text" href="https://github.com/codelion/optillm">"OptiLLM: Optimizing inference proxy for LLMs"</a>. <i>GitHub</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-08-05</span></span>.</cite></span>
</li>
<li id="cite_note-74"><span class="mw-cite-backlink"><b><a href="#cite_ref-74">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.marktechpost.com/2024/11/18/optillm-an-openai-api-compatible-optimizing-inference-proxy-which-implements-several-state-of-the-art-techniques-that-can-improve-the-accuracy-and-performance-of-llms/">"OptiLLM: An OpenAI API Compatible Optimizing Inference Proxy which Implements Several State-of-the-Art Techniques that can Improve the Accuracy and Performance of LLMs"</a>. <i>MarkTechPost</i>. 2024-11-18<span class="reference-accessdate">. Retrieved <span class="nowrap">2025-08-05</span></span>.</cite></span>
</li>
<li id="cite_note-75"><span class="mw-cite-backlink"><b><a href="#cite_ref-75">^</a></b></span> <span class="reference-text"><cite id="CITEREFKirosSalakhutdinovZemel2014" class="citation journal cs1">Kiros, Ryan; Salakhutdinov, Ruslan; Zemel, Rich (2014-06-18). <a rel="nofollow" class="external text" href="https://proceedings.mlr.press/v32/kiros14.html">"Multimodal Neural Language Models"</a>. <i>Proceedings of the 31st International Conference on Machine Learning</i>. PMLR: <span class="nowrap">595–</span>603. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230702195952/https://proceedings.mlr.press/v32/kiros14.html">Archived</a> from the original on 2023-07-02<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-02</span></span>.</cite></span>
</li>
<li id="cite_note-76"><span class="mw-cite-backlink"><b><a href="#cite_ref-76">^</a></b></span> <span class="reference-text"><cite id="CITEREFDriessXiaSajjadiLynch2023" class="citation arxiv cs1">Driess, Danny; Xia, Fei; Sajjadi, Mehdi S. M.; Lynch, Corey; Chowdhery, Aakanksha; Ichter, Brian; Wahid, Ayzaan; Tompson, Jonathan; Vuong, Quan; Yu, Tianhe; Huang, Wenlong; Chebotar, Yevgen; Sermanet, Pierre; Duckworth, Daniel; Levine, Sergey (2023-03-01). "PaLM-E: An Embodied Multimodal Language Model". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.03378">2303.03378</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-77"><span class="mw-cite-backlink"><b><a href="#cite_ref-77">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiuLiWuLee2023" class="citation arxiv cs1">Liu, Haotian; Li, Chunyuan; Wu, Qingyang; Lee, Yong Jae (2023-04-01). "Visual Instruction Tuning". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2304.08485">2304.08485</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-78"><span class="mw-cite-backlink"><b><a href="#cite_ref-78">^</a></b></span> <span class="reference-text"><cite id="CITEREFZhangLiBing2023" class="citation arxiv cs1">Zhang, Hang; Li, Xin; Bing, Lidong (2023-06-01). "Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2306.02858">2306.02858</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-79"><span class="mw-cite-backlink"><b><a href="#cite_ref-79">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.theregister.com/2024/05/13/openai_gpt4o/">"OpenAI says natively multimodal GPT-4o eats text, visuals, sound – and emits the same"</a>. <i>The Register</i>. 2024-05-13.</cite></span>
</li>
<li id="cite_note-80"><span class="mw-cite-backlink"><b><a href="#cite_ref-80">^</a></b></span> <span class="reference-text"><cite id="CITEREFZia2024" class="citation web cs1">Zia, Dr Tehseen (2024-01-08). <a rel="nofollow" class="external text" href="https://www.unite.ai/unveiling-of-large-multimodal-models-shaping-the-landscape-of-language-models-in-2024/">"Unveiling of Large Multimodal Models: Shaping the Landscape of Language Models in 2024"</a>. <i>Unite.AI</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-05-30</span></span>.</cite></span>
</li>
<li id="cite_note-81"><span class="mw-cite-backlink"><b><a href="#cite_ref-81">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiLiSavareseHoi2023" class="citation arxiv cs1">Li, Junnan; Li, Dongxu; Savarese, Silvio; Hoi, Steven (2023-01-01). "BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2301.12597">2301.12597</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-82"><span class="mw-cite-backlink"><b><a href="#cite_ref-82">^</a></b></span> <span class="reference-text"><cite id="CITEREFAlayracDonahueLucMiech2022" class="citation journal cs1">Alayrac, Jean-Baptiste; Donahue, Jeff; Luc, Pauline; Miech, Antoine; Barr, Iain; Hasson, Yana; Lenc, Karel; Mensch, Arthur; Millican, Katherine; Reynolds, Malcolm; Ring, Roman; Rutherford, Eliza; Cabi, Serkan; Han, Tengda; Gong, Zhitao (2022-12-06). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper_files/paper/2022/hash/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html">"Flamingo: a Visual Language Model for Few-Shot Learning"</a>. <i>Advances in Neural Information Processing Systems</i>. <b>35</b>: <span class="nowrap">23716–</span>23736. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2204.14198">2204.14198</a></span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230702195951/https://proceedings.neurips.cc/paper_files/paper/2022/hash/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html">Archived</a> from the original on 2023-07-02<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-02</span></span>.</cite></span>
</li>
<li id="cite_note-83"><span class="mw-cite-backlink"><b><a href="#cite_ref-83">^</a></b></span> <span class="reference-text"><cite id="CITEREFFinnie-AnsleyDennyBeckerLuxton-Reilly2022" class="citation book cs1">Finnie-Ansley, James; Denny, Paul; Becker, Brett A.; Luxton-Reilly, Andrew; Prather, James (14 February 2022). "The Robots Are Coming: Exploring the Implications of OpenAI Codex on Introductory Programming". <i>Proceedings of the 24th Australasian Computing Education Conference</i>. New York, NY, USA: Association for Computing Machinery. pp.&nbsp;<span class="nowrap">10–</span>19. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3511861.3511863">10.1145/3511861.3511863</a></span>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>978-1-4503-9643-1</bdi>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:246681316">246681316</a>.</cite></span>
</li>
<li id="cite_note-84"><span class="mw-cite-backlink"><b><a href="#cite_ref-84">^</a></b></span> <span class="reference-text"><cite id="CITEREFHuseinAburajouhCatal2025" class="citation journal cs1">Husein, Rasha Ahmad; Aburajouh, Hala; Catal, Cagatay (March 2025). "Large language models for code completion: A systematic literature review". <i>Computer Standards &amp; Interfaces</i>. <b>92</b> 103917. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1016%2Fj.csi.2024.103917">10.1016/j.csi.2024.103917</a>.</cite></span>
</li>
<li id="cite_note-85"><span class="mw-cite-backlink"><b><a href="#cite_ref-85">^</a></b></span> <span class="reference-text"><cite id="CITEREFWeissenowRost2025" class="citation journal cs1">Weissenow, Konstantin; Rost, Burkhard (April 2025). "Are protein language models the new universal key?". <i>Current Opinion in Structural Biology</i>. <b>91</b> 102997. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1016%2Fj.sbi.2025.102997">10.1016/j.sbi.2025.102997</a>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/39921962">39921962</a>.</cite></span>
</li>
<li id="cite_note-86"><span class="mw-cite-backlink"><b><a href="#cite_ref-86">^</a></b></span> <span class="reference-text"><cite id="CITEREFLinAkinRaoHie2023" class="citation journal cs1">Lin, Zeming; Akin, Halil; Rao, Roshan; Hie, Brian; Zhu, Zhongkai; Lu, Wenting; Smetanin, Nikita; Verkuil, Robert; Kabeli, Ori; Shmueli, Yaniv; dos Santos Costa, Allan; Fazel-Zarandi, Maryam; Sercu, Tom; Candido, Salvatore; Rives, Alexander (17 March 2023). <a rel="nofollow" class="external text" href="https://doi.org/10.1126%2Fscience.ade2574">"Evolutionary-scale prediction of atomic-level protein structure with a language model"</a>. <i>Science</i>. <b>379</b> (6637): <span class="nowrap">1123–</span>1130. <a href="Bibcode_(identifier)" class="mw-redirect" title="Bibcode (identifier)">Bibcode</a>:<a rel="nofollow" class="external text" href="https://ui.adsabs.harvard.edu/abs/2023Sci...379.1123L">2023Sci...379.1123L</a>. <a href="BioRxiv_(identifier)" class="mw-redirect" title="BioRxiv (identifier)">bioRxiv</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1101%2F2022.07.20.500902">10.1101/2022.07.20.500902</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1126%2Fscience.ade2574">10.1126/science.ade2574</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/36927031">36927031</a>.</cite></span>
</li>
<li id="cite_note-87"><span class="mw-cite-backlink"><b><a href="#cite_ref-87">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://esmatlas.com/about">"ESM Metagenomic Atlas | Meta AI"</a>. <i>esmatlas.com</i>.</cite></span>
</li>
<li id="cite_note-88"><span class="mw-cite-backlink"><b><a href="#cite_ref-88">^</a></b></span> <span class="reference-text"><cite id="CITEREFHayesRaoAkinSofroniew2025" class="citation journal cs1">Hayes, Thomas; Rao, Roshan; Akin, Halil; Sofroniew, Nicholas J.; Oktay, Deniz; Lin, Zeming; Verkuil, Robert; Tran, Vincent Q.; Deaton, Jonathan; Wiggert, Marius; Badkundri, Rohil; Shafkat, Irhum; Gong, Jun; Derry, Alexander; Molina, Raul S.; Thomas, Neil; Khan, Yousuf A.; Mishra, Chetan; Kim, Carolyn; Bartie, Liam J.; Nemeth, Matthew; Hsu, Patrick D.; Sercu, Tom; Candido, Salvatore; Rives, Alexander (21 February 2025). "Simulating 500 million years of evolution with a language model". <i>Science</i>. <b>387</b> (6736): <span class="nowrap">850–</span>858. <a href="Bibcode_(identifier)" class="mw-redirect" title="Bibcode (identifier)">Bibcode</a>:<a rel="nofollow" class="external text" href="https://ui.adsabs.harvard.edu/abs/2025Sci...387..850H">2025Sci...387..850H</a>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1126%2Fscience.ads0018">10.1126/science.ads0018</a>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/39818825">39818825</a>.</cite></span>
</li>
<li id="cite_note-89"><span class="mw-cite-backlink"><b><a href="#cite_ref-89">^</a></b></span> <span class="reference-text"><cite id="CITEREFFishmanKuratovShmelevPetrov2025" class="citation journal cs1">Fishman, Veniamin; Kuratov, Yuri; Shmelev, Aleksei; Petrov, Maxim; Penzar, Dmitry; Shepelin, Denis; Chekanov, Nikolay; Kardymon, Olga; Burtsev, Mikhail (11 January 2025). <a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11734698">"GENA-LM: a family of open-source foundational DNA language models for long sequences"</a>. <i>Nucleic Acids Research</i>. <b>53</b> (2): gkae1310. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1093%2Fnar%2Fgkae1310">10.1093/nar/gkae1310</a>. <a href="PMC_(identifier)" class="mw-redirect" title="PMC (identifier)">PMC</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11734698">11734698</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/39817513">39817513</a>.</cite></span>
</li>
<li id="cite_note-90"><span class="mw-cite-backlink"><b><a href="#cite_ref-90">^</a></b></span> <span class="reference-text"><cite id="CITEREFWangBianLiLi2024" class="citation journal cs1">Wang, Ning; Bian, Jiang; Li, Yuchen; Li, Xuhong; Mumtaz, Shahid; Kong, Linghe; Xiong, Haoyi (13 May 2024). <a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs42256-024-00836-4">"Multi-purpose RNA language modelling with motif-aware pretraining and type-guided fine-tuning"</a>. <i>Nature Machine Intelligence</i>. <b>6</b> (5): <span class="nowrap">548–</span>557. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs42256-024-00836-4">10.1038/s42256-024-00836-4</a></span>.</cite></span>
</li>
<li id="cite_note-fJta3-91"><span class="mw-cite-backlink"><b><a href="#cite_ref-fJta3_91-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFHoffmannBorgeaudMenschBuchatskaya2022" class="citation arxiv cs1">Hoffmann, Jordan; Borgeaud, Sebastian; Mensch, Arthur; Buchatskaya, Elena; Cai, Trevor; Rutherford, Eliza; Casas, Diego de Las; Hendricks, Lisa Anne; Welbl, Johannes; Clark, Aidan; Hennigan, Tom; Noland, Eric; Millican, Katie; Driessche, George van den; Damoc, Bogdan (2022-03-29). "Training Compute-Optimal Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2203.15556">2203.15556</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-IYm4Q-92"><span class="mw-cite-backlink">^ <a href="#cite_ref-IYm4Q_92-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-IYm4Q_92-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFCaballeroGuptaRishKrueger2022" class="citation arxiv cs1">Caballero, Ethan; Gupta, Kshitij; Rish, Irina; Krueger, David (2022). "Broken Neural Scaling Laws". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2210.14891">2210.14891</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-emergentpaper-93"><span class="mw-cite-backlink">^ <a href="#cite_ref-emergentpaper_93-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-emergentpaper_93-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFWeiTayBommasaniRaffel2022" class="citation journal cs1">Wei, Jason; Tay, Yi; Bommasani, Rishi; Raffel, Colin; Zoph, Barret; Borgeaud, Sebastian; Yogatama, Dani; Bosma, Maarten; Zhou, Denny; Metzler, Donald; Chi, Ed H.; Hashimoto, Tatsunori; Vinyals, Oriol; Liang, Percy; Dean, Jeff; Fedus, William (31 August 2022). <a rel="nofollow" class="external text" href="https://openreview.net/forum?id=yzkSU5zdwD">"Emergent Abilities of Large Language Models"</a>. <i>Transactions on Machine Learning Research</i>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/2835-8856">2835-8856</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230322210052/https://openreview.net/forum?id=yzkSU5zdwD">Archived</a> from the original on 22 March 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">19 March</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-JM6s1-94"><span class="mw-cite-backlink"><b><a href="#cite_ref-JM6s1_94-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.jasonwei.net/blog/emergence">"137 emergent abilities of large language models"</a>. <i>Jason Wei</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-24</span></span>.</cite></span>
</li>
<li id="cite_note-Bowman-95"><span class="mw-cite-backlink"><b><a href="#cite_ref-Bowman_95-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFBowman2023" class="citation arxiv cs1">Bowman, Samuel R. (2023). "Eight Things to Know about Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2304.00612">2304.00612</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-Hahn_20230314-96"><span class="mw-cite-backlink"><b><a href="#cite_ref-Hahn_20230314_96-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFHahnGoyal2023" class="citation arxiv cs1">Hahn, Michael; Goyal, Navin (2023-03-14). "A Theory of Emergent In-Context Learning as Implicit Structure Induction". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.07971">2303.07971</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-57FEA-97"><span class="mw-cite-backlink"><b><a href="#cite_ref-57FEA_97-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFPilehvarCamacho-Collados2019" class="citation journal cs1">Pilehvar, Mohammad Taher; Camacho-Collados, Jose (June 2019). <span class="id-lock-subscription" title="Paid subscription required"><a rel="nofollow" class="external text" href="https://aclanthology.org/N19-1128">"Proceedings of the 2019 Conference of the North"</a></span>. <i>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)</i>. Minneapolis, Minnesota: Association for Computational Linguistics: <span class="nowrap">1267–</span>1273. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.18653%2Fv1%2FN19-1128">10.18653/v1/N19-1128</a></span>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:102353817">102353817</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230627202732/https://aclanthology.org/N19-1128/">Archived</a> from the original on 2023-06-27<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-27</span></span>.</cite></span>
</li>
<li id="cite_note-TEIkA-98"><span class="mw-cite-backlink"><b><a href="#cite_ref-TEIkA_98-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://pilehvar.github.io/wic/">"WiC: The Word-in-Context Dataset"</a>. <i>pilehvar.github.io</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230627202725/https://pilehvar.github.io/wic/">Archived</a> from the original on 2023-06-27<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-27</span></span>.</cite></span>
</li>
<li id="cite_note-zgy1i-99"><span class="mw-cite-backlink"><b><a href="#cite_ref-zgy1i_99-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFPatelPavlick2021" class="citation journal cs1">Patel, Roma; Pavlick, Ellie (2021-10-06). <a rel="nofollow" class="external text" href="https://openreview.net/forum?id=gJcEM8sxHK">"Mapping Language Models to Grounded Conceptual Spaces"</a>. <i>ICLR</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230624191940/https://openreview.net/forum?id=gJcEM8sxHK">Archived</a> from the original on 2023-06-24<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-06-27</span></span>.</cite></span>
</li>
<li id="cite_note-Imb98-100"><span class="mw-cite-backlink"><b><a href="#cite_ref-Imb98_100-0">^</a></b></span> <span class="reference-text"><i><a rel="nofollow" class="external text" href="https://www.notion.so/A-Closer-Look-at-Large-Language-Models-Emergent-Abilities-493876b55df5479d80686f68a1abd72f">A Closer Look at Large Language Models Emergent Abilities</a> <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230624012329/https://www.notion.so/A-Closer-Look-at-Large-Language-Models-Emergent-Abilities-493876b55df5479d80686f68a1abd72f">Archived</a> 2023-06-24 at the <a href="Wayback_Machine" title="Wayback Machine">Wayback Machine</a></i> (Yao Fu, Nov 20, 2022)</span>
</li>
<li id="cite_note-CeQVF-101"><span class="mw-cite-backlink"><b><a href="#cite_ref-CeQVF_101-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFOrnes2023" class="citation web cs1">Ornes, Stephen (March 16, 2023). <a rel="nofollow" class="external text" href="https://www.quantamagazine.org/the-unpredictable-abilities-emerging-from-large-ai-models-20230316/">"The Unpredictable Abilities Emerging From Large AI Models"</a>. <i>Quanta Magazine</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230316203438/https://www.quantamagazine.org/the-unpredictable-abilities-emerging-from-large-ai-models-20230316/">Archived</a> from the original on March 16, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">March 16,</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-C775b-102"><span class="mw-cite-backlink"><b><a href="#cite_ref-C775b_102-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFSchaefferMirandaKoyejo2023" class="citation arxiv cs1">Schaeffer, Rylan; Miranda, Brando; Koyejo, Sanmi (2023-04-01). "Are Emergent Abilities of Large Language Models a Mirage?". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2304.15004">2304.15004</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-103"><span class="mw-cite-backlink"><b><a href="#cite_ref-103">^</a></b></span> <span class="reference-text"><cite id="CITEREFBlank2023" class="citation journal cs1">Blank, Idan A. (November 2023). <a rel="nofollow" class="external text" href="https://doi.org/10.1016%2Fj.tics.2023.08.006">"What are large language models supposed to model?"</a>. <i>Trends in Cognitive Sciences</i>. <b>27</b> (11): <span class="nowrap">987–</span>989. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1016%2Fj.tics.2023.08.006">10.1016/j.tics.2023.08.006</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/37659920">37659920</a>.</cite></span>
</li>
<li id="cite_note-oYGlo-104"><span class="mw-cite-backlink"><b><a href="#cite_ref-oYGlo_104-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFNandaChanLieberumSmith2023" class="citation arxiv cs1">Nanda, Neel; Chan, Lawrence; Lieberum, Tom; Smith, Jess; Steinhardt, Jacob (2023-01-01). "Progress measures for grokking via mechanistic interpretability". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2301.05217">2301.05217</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-105"><span class="mw-cite-backlink"><b><a href="#cite_ref-105">^</a></b></span> <span class="reference-text"><cite id="CITEREFAnanthaswamy2024" class="citation web cs1">Ananthaswamy, Anil (2024-04-12). <a rel="nofollow" class="external text" href="https://www.quantamagazine.org/how-do-machines-grok-data-20240412/">"How Do Machines 'Grok' Data?"</a>. <i>Quanta Magazine</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-06-30</span></span>.</cite></span>
</li>
<li id="cite_note-106"><span class="mw-cite-backlink"><b><a href="#cite_ref-106">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://transformer-circuits.pub/2025/attribution-graphs/biology.html#dives-poems%7Ctitle=On">"On the Biology of a Large Language Model"</a>. <i>Transformer Circuits</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-06-30</span></span>.</cite></span>
</li>
<li id="cite_note-debate_understanding-107"><span class="mw-cite-backlink">^ <a href="#cite_ref-debate_understanding_107-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-debate_understanding_107-1"><sup><i><b>b</b></i></sup></a> <a href="#cite_ref-debate_understanding_107-2"><sup><i><b>c</b></i></sup></a> <a href="#cite_ref-debate_understanding_107-3"><sup><i><b>d</b></i></sup></a> <a href="#cite_ref-debate_understanding_107-4"><sup><i><b>e</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFMitchellKrakauer2023" class="citation journal cs1">Mitchell, Melanie; Krakauer, David C. (28 March 2023). <a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10068812">"The debate over understanding in AI's large language models"</a>. <i>Proceedings of the National Academy of Sciences</i>. <b>120</b> (13): e2215907120. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2210.13966">2210.13966</a></span>. <a href="Bibcode_(identifier)" class="mw-redirect" title="Bibcode (identifier)">Bibcode</a>:<a rel="nofollow" class="external text" href="https://ui.adsabs.harvard.edu/abs/2023PNAS..12015907M">2023PNAS..12015907M</a>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1073%2Fpnas.2215907120">10.1073/pnas.2215907120</a></span>. <a href="PMC_(identifier)" class="mw-redirect" title="PMC (identifier)">PMC</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10068812">10068812</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/36943882">36943882</a>.</cite></span>
</li>
<li id="cite_note-O8Upd-108"><span class="mw-cite-backlink"><b><a href="#cite_ref-O8Upd_108-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFMetz2023" class="citation news cs1">Metz, Cade (16 May 2023). <a rel="nofollow" class="external text" href="https://www.nytimes.com/2023/05/16/technology/microsoft-ai-human-reasoning.html">"Microsoft Says New A.I. Shows Signs of Human Reasoning"</a>. <i>The New York Times</i>.</cite></span>
</li>
<li id="cite_note-microsoft_sparks-109"><span class="mw-cite-backlink">^ <a href="#cite_ref-microsoft_sparks_109-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-microsoft_sparks_109-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFBubeckChandrasekaranEldanGehrke2023" class="citation arxiv cs1">Bubeck, Sébastien; Chandrasekaran, Varun; Eldan, Ronen; Gehrke, Johannes; Horvitz, Eric; Kamar, Ece; Lee, Peter; Lee, Yin Tat; Li, Yuanzhi; Lundberg, Scott; Nori, Harsha; Palangi, Hamid; Ribeiro, Marco Tulio; Zhang, Yi (2023). "Sparks of Artificial General Intelligence: Early experiments with GPT-4". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.12712">2303.12712</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-110"><span class="mw-cite-backlink"><b><a href="#cite_ref-110">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.fastcompany.com/91211163/anthropic-ceo-dario-amodei-pens-a-smart-look-at-our-ai-future">"Anthropic CEO Dario Amodei pens a smart look at our AI future"</a>. <i>Fast Company</i>. October 17, 2024.</cite></span>
</li>
<li id="cite_note-rEEmH-111"><span class="mw-cite-backlink"><b><a href="#cite_ref-rEEmH_111-0">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.zdnet.com/article/chatgpt-is-more-like-an-alien-intelligence-than-a-human-brain-says-futurist/">"ChatGPT is more like an 'alien intelligence' than a human brain, says futurist"</a>. <i>ZDNET</i>. 2023. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230612065937/https://www.zdnet.com/article/chatgpt-is-more-like-an-alien-intelligence-than-a-human-brain-says-futurist/">Archived</a> from the original on 12 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">12 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-new_yorker_kind_of_mind-112"><span class="mw-cite-backlink">^ <a href="#cite_ref-new_yorker_kind_of_mind_112-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-new_yorker_kind_of_mind_112-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFNewport2023" class="citation magazine cs1">Newport, Cal (13 April 2023). <a rel="nofollow" class="external text" href="https://www.newyorker.com/science/annals-of-artificial-intelligence/what-kind-of-mind-does-chatgpt-have">"What Kind of Mind Does ChatGPT Have?"</a>. <i>The New Yorker</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230612071443/https://www.newyorker.com/science/annals-of-artificial-intelligence/what-kind-of-mind-does-chatgpt-have">Archived</a> from the original on 12 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">12 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-rAFIZ-113"><span class="mw-cite-backlink"><b><a href="#cite_ref-rAFIZ_113-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFRoose2023" class="citation news cs1">Roose, Kevin (30 May 2023). <a rel="nofollow" class="external text" href="https://www.nytimes.com/2023/05/30/technology/shoggoth-meme-ai.html">"Why an Octopus-like Creature Has Come to Symbolize the State of A.I."</a> <i>The New York Times</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230530193814/https://www.nytimes.com/2023/05/30/technology/shoggoth-meme-ai.html">Archived</a> from the original on 30 May 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">12 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-4luKE-114"><span class="mw-cite-backlink"><b><a href="#cite_ref-4luKE_114-0">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://time.com/6271657/a-to-z-of-artificial-intelligence/">"The A to Z of Artificial Intelligence"</a>. <i>Time Magazine</i>. 13 April 2023. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230616123839/https://time.com/6271657/a-to-z-of-artificial-intelligence/">Archived</a> from the original on 16 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">12 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-hallucination-survey-115"><span class="mw-cite-backlink"><b><a href="#cite_ref-hallucination-survey_115-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFJiLeeFrieskeYu2022" class="citation journal cs1">Ji, Ziwei; Lee, Nayeon; Frieske, Rita; Yu, Tiezheng; Su, Dan; Xu, Yan; Ishii, Etsuko; Bang, Yejin; Dai, Wenliang; Madotto, Andrea; Fung, Pascale (November 2022). <a rel="nofollow" class="external text" href="https://dl.acm.org/doi/pdf/10.1145/3571730">"Survey of Hallucination in Natural Language Generation"</a> <span class="cs1-format">(pdf)</span>. <i>ACM Computing Surveys</i>. <b>55</b> (12). <a href="Association_for_Computing_Machinery" title="Association for Computing Machinery">Association for Computing Machinery</a>: <span class="nowrap">1–</span>38. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2202.03629">2202.03629</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3571730">10.1145/3571730</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:246652372">246652372</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230326145635/https://dl.acm.org/doi/pdf/10.1145/3571730">Archived</a> from the original on 26 March 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">15 January</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-116"><span class="mw-cite-backlink"><b><a href="#cite_ref-116">^</a></b></span> <span class="reference-text"><cite id="CITEREFVarshneyYaoZhangChen2023" class="citation arxiv cs1">Varshney, Neeraj; Yao, Wenlin; Zhang, Hongming; Chen, Jianshu; Yu, Dong (2023). "A Stitch in Time Saves Nine: Detecting and Mitigating Hallucinations of LLMs by Validating Low-Confidence Generation". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2307.03987">2307.03987</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-Lin-2025-02-05-WSJ-117"><span class="mw-cite-backlink"><b><a href="#cite_ref-Lin-2025-02-05-WSJ_117-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLin2025" class="citation journal cs1">Lin, Belle (2025-02-05). <a rel="nofollow" class="external text" href="https://www.wsj.com/articles/why-amazon-is-betting-on-automated-reasoning-to-reduce-ais-hallucinations-b838849e">"Why Amazon is Betting on 'Automated Reasoning' to Reduce AI's Hallucinations: The tech giant says an obscure field that combines AI and math can mitigate—but not completely eliminate—AI's propensity to provide wrong answers"</a>. <i>Wall Street Journal</i>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/0099-9660">0099-9660</a>.</cite></span>
</li>
<li id="cite_note-118"><span class="mw-cite-backlink"><b><a href="#cite_ref-118">^</a></b></span> <span class="reference-text"><cite id="CITEREFLakoff1999" class="citation book cs1">Lakoff, George (1999). <i>Philosophy in the Flesh: The Embodied Mind and Its Challenge to Western Philosophy; Appendix: The Neural Theory of Language Paradigm</i>. New York Basic Books. pp.&nbsp;<span class="nowrap">569–</span>583. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>978-0-465-05674-3</bdi>.</cite></span>
</li>
<li id="cite_note-119"><span class="mw-cite-backlink"><b><a href="#cite_ref-119">^</a></b></span> <span class="reference-text"><cite id="CITEREFEvans2014" class="citation book cs1">Evans, Vyvyan. (2014). <i>The Language Myth</i>. Cambridge University Press. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>978-1-107-04396-1</bdi>.</cite></span>
</li>
<li id="cite_note-120"><span class="mw-cite-backlink"><b><a href="#cite_ref-120">^</a></b></span> <span class="reference-text"><cite id="CITEREFFriston2022" class="citation book cs1">Friston, Karl J. (2022). <i>Active Inference: The Free Energy Principle in Mind, Brain, and Behavior; Chapter 4 The Generative Models of Active Inference</i>. The MIT Press. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>978-0-262-36997-8</bdi>.</cite></span>
</li>
<li id="cite_note-few-shot-learners-121"><span class="mw-cite-backlink">^ <a href="#cite_ref-few-shot-learners_121-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-few-shot-learners_121-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFBrownMannRyderSubbiah2020" class="citation journal cs1">Brown, Tom B.; Mann, Benjamin; Ryder, Nick; Subbiah, Melanie; Kaplan, Jared; Dhariwal, Prafulla; Neelakantan, Arvind; Shyam, Pranav; Sastry, Girish; Askell, Amanda; Agarwal, Sandhini; Herbert-Voss, Ariel; Krueger, Gretchen; Henighan, Tom; Child, Rewon; Ramesh, Aditya; Ziegler, Daniel M.; Wu, Jeffrey; Winter, Clemens; Hesse, Christopher; Chen, Mark; Sigler, Eric; Litwin, Mateusz; Gray, Scott; Chess, Benjamin; Clark, Jack; Berner, Christopher; McCandlish, Sam; Radford, Alec; Sutskever, Ilya; Amodei, Dario (Dec 2020). Larochelle, H.; Ranzato, M.; Hadsell, R.; Balcan, M.F.; Lin, H. (eds.). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">"Language Models are Few-Shot Learners"</a> <span class="cs1-format">(PDF)</span>. <i>Advances in Neural Information Processing Systems</i>. <b>33</b>. Curran Associates, Inc.: <span class="nowrap">1877–</span>1901. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231117204007/https://proceedings.neurips.cc/paper/2020/file/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2023-11-17<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-03-14</span></span>.</cite></span>
</li>
<li id="cite_note-Huyen-122"><span class="mw-cite-backlink">^ <a href="#cite_ref-Huyen_122-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-Huyen_122-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFHuyen2019" class="citation web cs1">Huyen, Chip (October 18, 2019). <a rel="nofollow" class="external text" href="https://thegradient.pub/understanding-evaluation-metrics-for-language-models/">"Evaluation Metrics for Language Modeling"</a>. <i>The Gradient</i><span class="reference-accessdate">. Retrieved <span class="nowrap">January 14,</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-123"><span class="mw-cite-backlink"><b><a href="#cite_ref-123">^</a></b></span> <span class="reference-text"><cite id="CITEREFEdwards2023" class="citation web cs1">Edwards, Benj (2023-09-28). <a rel="nofollow" class="external text" href="https://arstechnica.com/information-technology/2023/09/ai-language-models-can-exceed-png-and-flac-in-lossless-compression-says-study/">"AI language models can exceed PNG and FLAC in lossless compression, says study"</a>. <i>Ars Technica</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-05-29</span></span>.</cite></span>
</li>
<li id="cite_note-124"><span class="mw-cite-backlink"><b><a href="#cite_ref-124">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://github.com/openai/simple-evals">"openai/simple-evals"</a>. OpenAI. 2024-05-28<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-05-28</span></span>.</cite></span>
</li>
<li id="cite_note-125"><span class="mw-cite-backlink"><b><a href="#cite_ref-125">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://github.com/openai/evals">"openai/evals"</a>. OpenAI. 2024-05-28. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240508225708/https://github.com/openai/evals">Archived</a> from the original on 2024-05-08<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-05-28</span></span>.</cite></span>
</li>
<li id="cite_note-boolq-126"><span class="mw-cite-backlink">^ <a href="#cite_ref-boolq_126-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-boolq_126-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFClarkLeeChangKwiatkowski2019" class="citation arxiv cs1">Clark, Christopher; Lee, Kenton; Chang, Ming-Wei; Kwiatkowski, Tom; Collins, Michael; Toutanova, Kristina (2019). "BoolQ: Exploring the Surprising Difficulty of Natural Yes/No Questions". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1905.10044">1905.10044</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-survey-127"><span class="mw-cite-backlink">^ <a href="#cite_ref-survey_127-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-survey_127-1"><sup><i><b>b</b></i></sup></a> <a href="#cite_ref-survey_127-2"><sup><i><b>c</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFWayne_Xin_ZhaoZhouLiTang2023" class="citation arxiv cs1">Wayne Xin Zhao; Zhou, Kun; Li, Junyi; Tang, Tianyi; Wang, Xiaolei; Hou, Yupeng; Min, Yingqian; Zhang, Beichen; Zhang, Junjie; Dong, Zican; Du, Yifan; Yang, Chen; Chen, Yushuo; Chen, Zhipeng; Jiang, Jinhao; Ren, Ruiyang; Li, Yifan; Tang, Xinyu; Liu, Zikang; Liu, Peiyu; Nie, Jian-Yun; Wen, Ji-Rong (2023). "A Survey of Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.18223">2303.18223</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-128"><span class="mw-cite-backlink"><b><a href="#cite_ref-128">^</a></b></span> <span class="reference-text"><cite id="CITEREFNangia,_Nikita_and_Vania,_Clara_and_Bhalerao,_Rasika_and_Bowman,_Samuel_R.2020" class="citation conference cs1">Nangia, Nikita and Vania, Clara and Bhalerao, Rasika and Bowman, Samuel R. (November 2020). <a rel="nofollow" class="external text" href="https://aclanthology.org/2020.emnlp-main.154/">"CrowS-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models"</a>. In Webber, Bonnie and Cohn, Trevor and He, Yulan and Liu, Yang (ed.). <i>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)</i>. Association for Computational Linguistics. pp.&nbsp;<span class="nowrap">1953–</span>1967. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2010.00133">2010.00133</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.18653%2Fv1%2F2020.emnlp-main.154">10.18653/v1/2020.emnlp-main.154</a>.</cite><span class="cs1-maint citation-comment"><code class="cs1-code">{{cite conference}}</code>: CS1 maint: multiple names: authors list (link)</span></span>
</li>
<li id="cite_note-129"><span class="mw-cite-backlink"><b><a href="#cite_ref-129">^</a></b></span> <span class="reference-text"><cite id="CITEREFNadeem,_Moin_and_Bethke,_Anna_and_Reddy,_Siva2021" class="citation conference cs1">Nadeem, Moin and Bethke, Anna and Reddy, Siva (August 2021). <a rel="nofollow" class="external text" href="https://aclanthology.org/2021.acl-long.416/">"StereoSet: Measuring stereotypical bias in pretrained language models"</a>. In Zong, Chengqing and Xia, Fei and Li, Wenjie and Navigli, Roberto (ed.). <i>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</i>. Association for Computational Linguistics. pp.&nbsp;<span class="nowrap">5356–</span>5371. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2004.09456">2004.09456</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.18653%2Fv1%2F2021.acl-long.416">10.18653/v1/2021.acl-long.416</a>.</cite><span class="cs1-maint citation-comment"><code class="cs1-code">{{cite conference}}</code>: CS1 maint: multiple names: authors list (link)</span></span>
</li>
<li id="cite_note-130"><span class="mw-cite-backlink"><b><a href="#cite_ref-130">^</a></b></span> <span class="reference-text"><cite id="CITEREFSimpson,_Shmona_and_Nukpezah,_Jonathan_and_Kie_Brooks_and_Pandya,_Raaghav2024" class="citation journal cs1">Simpson, Shmona and Nukpezah, Jonathan and Kie Brooks and Pandya, Raaghav (17 December 2024). <a rel="nofollow" class="external text" href="https://doi.org/10.1007%2Fs43681-024-00613-4">"Parity benchmark for measuring bias in LLMs"</a>. <i>AI and Ethics</i>. <b>5</b> (3). Springer: <span class="nowrap">3087–</span>3101. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1007%2Fs43681-024-00613-4">10.1007/s43681-024-00613-4</a></span>.</cite><span class="cs1-maint citation-comment"><code class="cs1-code">{{cite journal}}</code>: CS1 maint: multiple names: authors list (link)</span></span>
</li>
<li id="cite_note-131"><span class="mw-cite-backlink"><b><a href="#cite_ref-131">^</a></b></span> <span class="reference-text"><cite id="CITEREFCaramancion2023" class="citation book cs1">Caramancion, Kevin Matthe (2023-11-13). "News Verifiers Showdown: A Comparative Performance Evaluation of ChatGPT 3.5, ChatGPT 4.0, Bing AI, and Bard in News Fact-Checking". <i>2023 IEEE Future Networks World Forum (FNWF)</i>. IEEE. pp.&nbsp;<span class="nowrap">1–</span>6. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2306.17176">2306.17176</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1109%2FFNWF58287.2023.10520446">10.1109/FNWF58287.2023.10520446</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>979-8-3503-2458-7</bdi>.</cite></span>
</li>
<li id="cite_note-132"><span class="mw-cite-backlink"><b><a href="#cite_ref-132">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://imbue.com/research/70b-evals/">"Sanitized open-source datasets for natural language and code understanding: how we evaluated our 70B model"</a>. <i>imbue.com</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240726173012/https://imbue.com/research/70b-evals/">Archived</a> from the original on 2024-07-26<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-07-24</span></span>.</cite></span>
</li>
<li id="cite_note-bigbench-133"><span class="mw-cite-backlink"><b><a href="#cite_ref-bigbench_133-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFSrivastavaRastogiRaoAbu_Awal_Md_Shoeb2022" class="citation arxiv cs1">Srivastava, Aarohi; et&nbsp;al. (2022). "Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2206.04615">2206.04615</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-truthfulqa-134"><span class="mw-cite-backlink"><b><a href="#cite_ref-truthfulqa_134-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLinHiltonEvans2021" class="citation arxiv cs1">Lin, Stephanie; Hilton, Jacob; Evans, Owain (2021). "TruthfulQA: Measuring How Models Mimic Human Falsehoods". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2109.07958">2109.07958</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-hellaswag-135"><span class="mw-cite-backlink">^ <a href="#cite_ref-hellaswag_135-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-hellaswag_135-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFZellersHoltzmanBiskFarhadi2019" class="citation arxiv cs1">Zellers, Rowan; Holtzman, Ari; Bisk, Yonatan; Farhadi, Ali; Choi, Yejin (2019). "HellaSwag: Can a Machine Really Finish Your Sentence?". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1905.07830">1905.07830</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-ZDTUM-136"><span class="mw-cite-backlink"><b><a href="#cite_ref-ZDTUM_136-0">^</a></b></span> <span class="reference-text"><cite class="citation journal cs1"><a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs41551-023-01012-6">"Prepare for truly useful large language models"</a>. <i>Nature Biomedical Engineering</i>. <b>7</b> (2): <span class="nowrap">85–</span>86. 7 March 2023. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs41551-023-01012-6">10.1038/s41551-023-01012-6</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/36882584">36882584</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:257403466">257403466</a>.</cite></span>
</li>
<li id="cite_note-81w7x-137"><span class="mw-cite-backlink"><b><a href="#cite_ref-81w7x_137-0">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.economist.com/finance-and-economics/2023/05/07/your-job-is-probably-safe-from-artificial-intelligence">"Your job is (probably) safe from artificial intelligence"</a>. <i>The Economist</i>. 7 May 2023. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230617225618/https://www.economist.com/finance-and-economics/2023/05/07/your-job-is-probably-safe-from-artificial-intelligence">Archived</a> from the original on 17 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">18 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-zIM6Y-138"><span class="mw-cite-backlink"><b><a href="#cite_ref-zIM6Y_138-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.goldmansachs.com/intelligence/pages/generative-ai-could-raise-global-gdp-by-7-percent.html">"Generative AI Could Raise Global GDP by 7%"</a>. <i>Goldman Sachs</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230618013836/https://www.goldmansachs.com/intelligence/pages/generative-ai-could-raise-global-gdp-by-7-percent.html">Archived</a> from the original on 18 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">18 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-139"><span class="mw-cite-backlink"><b><a href="#cite_ref-139">^</a></b></span> <span class="reference-text"><cite id="CITEREFBrinkmannBaumannBonnefonDerex2023" class="citation journal cs1">Brinkmann, Levin; Baumann, Fabian; Bonnefon, Jean-François; Derex, Maxime; Müller, Thomas F.; Nussberger, Anne-Marie; Czaplicka, Agnieszka; Acerbi, Alberto; Griffiths, Thomas L.; Henrich, Joseph; Leibo, Joel Z.; McElreath, Richard; Oudeyer, Pierre-Yves; Stray, Jonathan; Rahwan, Iyad (2023-11-20). <a rel="nofollow" class="external text" href="https://www.nature.com/articles/s41562-023-01742-2">"Machine culture"</a>. <i>Nature Human Behaviour</i>. <b>7</b> (11): <span class="nowrap">1855–</span>1868. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2311.11388">2311.11388</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs41562-023-01742-2">10.1038/s41562-023-01742-2</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/2397-3374">2397-3374</a>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/37985914">37985914</a>.</cite></span>
</li>
<li id="cite_note-140"><span class="mw-cite-backlink"><b><a href="#cite_ref-140">^</a></b></span> <span class="reference-text"><cite id="CITEREFPengWangDeng2023" class="citation journal cs1">Peng, Zhencan; Wang, Zhizhi; Deng, Dong (13 June 2023). <a rel="nofollow" class="external text" href="https://people.cs.rutgers.edu/~dd903/assets/papers/sigmod23.pdf">"Near-Duplicate Sequence Search at Scale for Large Language Model Memorization Evaluation"</a> <span class="cs1-format">(PDF)</span>. <i>Proceedings of the ACM on Management of Data</i>. <b>1</b> (2): <span class="nowrap">1–</span>18. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3589324">10.1145/3589324</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:259213212">259213212</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240827053753/https://people.cs.rutgers.edu/~dd903/assets/papers/sigmod23.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2024-08-27<span class="reference-accessdate">. Retrieved <span class="nowrap">2024-01-20</span></span>.</cite> Citing Lee et al 2022.</span>
</li>
<li id="cite_note-141"><span class="mw-cite-backlink"><b><a href="#cite_ref-141">^</a></b></span> <span class="reference-text"><a href="#CITEREFPengWangDeng2023">Peng, Wang &amp; Deng 2023</a>, p.&nbsp;8.</span>
</li>
<li id="cite_note-142"><span class="mw-cite-backlink"><b><a href="#cite_ref-142">^</a></b></span> <span class="reference-text"><cite id="CITEREFStephen_Council2023" class="citation web cs1">Stephen Council (1 Dec 2023). <a rel="nofollow" class="external text" href="https://www.sfgate.com/tech/article/google-openai-chatgpt-break-model-18525445.php">"How Googlers cracked an SF rival's tech model with a single word"</a>. SFGATE. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20231216160941/https://www.sfgate.com/tech/article/google-openai-chatgpt-break-model-18525445.php">Archived</a> from the original on 16 December 2023.</cite></span>
</li>
<li id="cite_note-nD6kH-143"><span class="mw-cite-backlink"><b><a href="#cite_ref-nD6kH_143-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFAlba2023" class="citation news cs1">Alba, Davey (1 May 2023). <a rel="nofollow" class="external text" href="https://www.japantimes.co.jp/news/2023/05/01/business/tech/ai-fake-news-content-farms/">"AI chatbots have been used to create dozens of news content farms"</a>. <i>The Japan Times</i><span class="reference-accessdate">. Retrieved <span class="nowrap">18 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-PKiPY-144"><span class="mw-cite-backlink"><b><a href="#cite_ref-PKiPY_144-0">^</a></b></span> <span class="reference-text"><cite class="citation journal cs1"><span class="id-lock-subscription" title="Paid subscription required"><a rel="nofollow" class="external text" href="https://www.science.org/content/article/could-chatbots-help-devise-next-pandemic-virus">"Could chatbots help devise the next pandemic virus?"</a></span>. <i>Science</i>. 14 June 2023. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1126%2Fscience.adj2463">10.1126/science.adj2463</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230618013834/https://www.science.org/content/article/could-chatbots-help-devise-next-pandemic-virus">Archived</a> from the original on 18 June 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">18 June</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-145"><span class="mw-cite-backlink"><b><a href="#cite_ref-145">^</a></b></span> <span class="reference-text"><cite id="CITEREFEdwards2024" class="citation web cs1">Edwards, Benj (2024-01-15). <a rel="nofollow" class="external text" href="https://arstechnica.com/information-technology/2024/01/ai-poisoning-could-turn-open-models-into-destructive-sleeper-agents-says-anthropic/">"AI poisoning could turn models into destructive "sleeper agents," says Anthropic"</a>. <i>Ars Technica</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-07-19</span></span>.</cite></span>
</li>
<li id="cite_note-146"><span class="mw-cite-backlink"><b><a href="#cite_ref-146">^</a></b></span> <span class="reference-text"><cite id="CITEREFKang2023" class="citation arxiv cs1">Kang, Daniel (2023). "Exploiting programmatic behavior of LLMs: Dual-use through standard security attacks". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2302.05733">2302.05733</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CR">cs.CR</a>].</cite></span>
</li>
<li id="cite_note-:2-147"><span class="mw-cite-backlink">^ <a href="#cite_ref-:2_147-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-:2_147-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://www.americansunlight.org/updates/new-report-russian-propaganda-may-be-flooding-ai-models">"Russian propaganda may be flooding AI models"</a>. <i>The American Sunlight Project</i>. 26 February 2025<span class="reference-accessdate">. Retrieved <span class="nowrap">2025-04-11</span></span>.</cite></span>
</li>
<li id="cite_note-148"><span class="mw-cite-backlink"><b><a href="#cite_ref-148">^</a></b></span> <span class="reference-text"><cite id="CITEREFGoudarzi2025" class="citation web cs1">Goudarzi, Sara (2025-03-26). <a rel="nofollow" class="external text" href="https://thebulletin.org/2025/03/russian-networks-flood-the-internet-with-propaganda-aiming-to-corrupt-ai-chatbots/">"Russian networks flood the Internet with propaganda, aiming to corrupt AI chatbots"</a>. <i><a href="Bulletin_of_the_Atomic_Scientists" title="Bulletin of the Atomic Scientists">Bulletin of the Atomic Scientists</a></i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-04-10</span></span>.</cite></span>
</li>
<li id="cite_note-149"><span class="mw-cite-backlink"><b><a href="#cite_ref-149">^</a></b></span> <span class="reference-text"><cite id="CITEREFWang2024" class="citation web cs1">Wang, Yongge (20 June 2024). <a rel="nofollow" class="external text" href="https://eprint.iacr.org/2024/586.pdf">"Encryption Based Covert Channel for Large Language Models"</a> <span class="cs1-format">(PDF)</span>. IACR ePrint 2024/586. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240624191233/https://eprint.iacr.org/2024/586.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 24 June 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">24 June</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-150"><span class="mw-cite-backlink"><b><a href="#cite_ref-150">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://github.com/openai/openai-python/blob/v0.27.6/chatml.md">"openai-python/chatml.md at v0.27.6 · openai/openai-python"</a>. <i>GitHub</i>.</cite></span>
</li>
<li id="cite_note-auto1-151"><span class="mw-cite-backlink"><b><a href="#cite_ref-auto1_151-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFDouglas2023" class="citation web cs1">Douglas, Will (March 3, 2023). <a rel="nofollow" class="external text" href="https://www.technologyreview.com/2023/03/03/1069311/inside-story-oral-history-how-chatgpt-built-openai/">"The inside story of how ChatGPT was built from the people who made it"</a>. <i>MIT Technology Review</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230303093219/https://www.technologyreview.com/2023/03/03/1069311/inside-story-oral-history-how-chatgpt-built-openai/">Archived</a> from the original on March 3, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">March 6,</span> 2023</span>.</cite></span>
</li>
<li id="cite_note-152"><span class="mw-cite-backlink"><b><a href="#cite_ref-152">^</a></b></span> <span class="reference-text"><cite id="CITEREFGreshakeAbdelnabiMishraEndres2023" class="citation arxiv cs1">Greshake, Kai; Abdelnabi, Sahar; Mishra, Shailesh; Endres, Christoph; Holz, Thorsten; Fritz, Mario (2023-02-01). "Not what you've signed up for: Compromising Real-World LLM-Integrated Applications with Indirect Prompt Injection". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2302.12173">2302.12173</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CR">cs.CR</a>].</cite></span>
</li>
<li id="cite_note-:8-153"><span class="mw-cite-backlink">^ <a href="#cite_ref-:8_153-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-:8_153-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFXuWangXueHu2025" class="citation arxiv cs1">Xu, Weijie; Wang, Yiwen; Xue, Chi; Hu, Xiangkun; Fang, Xi; Dong, Guimin; Reddy, Chandan K. (2025-06-28). "Quantifying Fairness in LLMs Beyond Tokens: A Semantic and Statistical Perspective". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2506.19028v1">2506.19028v1</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-:1-154"><span class="mw-cite-backlink"><b><a href="#cite_ref-:1_154-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFLuoPuettSmith2023" class="citation arxiv cs1">Luo, Queenie; Puett, Michael J.; Smith, Michael D. (2023-03-28). "A Perspectival Mirror of the Elephant: Investigating Language Bias on Google, ChatGPT, Wikipedia, and YouTube". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.16281v2">2303.16281v2</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CY">cs.CY</a>].</cite></span>
</li>
<li id="cite_note-155"><span class="mw-cite-backlink"><b><a href="#cite_ref-155">^</a></b></span> <span class="reference-text"><cite id="CITEREFWangMorgensternDickerson2025" class="citation journal cs1">Wang, Angelina; <a href="Jamie_Morgenstern" title="Jamie Morgenstern">Morgenstern, Jamie</a>; Dickerson, John P. (17 February 2025). "Large language models that replace human participants can harmfully misportray and flatten identity groups". <i>Nature Machine Intelligence</i>. <b>7</b> (3): <span class="nowrap">400–</span>411. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2402.01908">2402.01908</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs42256-025-00986-z">10.1038/s42256-025-00986-z</a>.</cite></span>
</li>
<li id="cite_note-156"><span class="mw-cite-backlink"><b><a href="#cite_ref-156">^</a></b></span> <span class="reference-text"><cite id="CITEREFChengDurmusJurafsky2023" class="citation arxiv cs1">Cheng, Myra; Durmus, Esin; Jurafsky, Dan (2023-05-29). "Marked Personas: Using Natural Language Prompts to Measure Stereotypes in Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2305.18189">2305.18189</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-157"><span class="mw-cite-backlink"><b><a href="#cite_ref-157">^</a></b></span> <span class="reference-text"><cite id="CITEREFKotekDockumSun2023" class="citation book cs1">Kotek, Hadas; Dockum, Rikker; Sun, David (2023-11-05). "Gender bias and stereotypes in Large Language Models". <a rel="nofollow" class="external text" href="https://dl.acm.org/doi/10.1145/3582269.3615599"><i>Proceedings of the ACM Collective Intelligence Conference</i></a>. New York, NY, USA: Association for Computing Machinery. pp.&nbsp;<span class="nowrap">12–</span>24. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2308.14921">2308.14921</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3582269.3615599">10.1145/3582269.3615599</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>979-8-4007-0113-9</bdi>.</cite></span>
</li>
<li id="cite_note-158"><span class="mw-cite-backlink"><b><a href="#cite_ref-158">^</a></b></span> <span class="reference-text"><cite id="CITEREFChoiXuXueEckman2024" class="citation arxiv cs1">Choi, Hyeong Kyu; Xu, Weijie; Xue, Chi; Eckman, Stephanie; Reddy, Chandan K. (2024-09-27). "Mitigating Selection Bias with Node Pruning and Auxiliary Options". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2409.18857">2409.18857</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-159"><span class="mw-cite-backlink"><b><a href="#cite_ref-159">^</a></b></span> <span class="reference-text"><cite id="CITEREFZhengZhouMengZhou2023" class="citation arxiv cs1">Zheng, Chujie; Zhou, Hao; Meng, Fandong; Zhou, Jie; Huang, Minlie (2023-09-07). "Large Language Models Are Not Robust Multiple Choice Selectors". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2309.03882">2309.03882</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-160"><span class="mw-cite-backlink"><b><a href="#cite_ref-160">^</a></b></span> <span class="reference-text"><cite id="CITEREFHeikkilä2023" class="citation web cs1">Heikkilä, Melissa (August 7, 2023). <a rel="nofollow" class="external text" href="https://www.technologyreview.com/2023/08/07/1077324/ai-language-models-are-rife-with-political-biases/">"AI language models are rife with different political biases"</a>. <i>MIT Technology Review</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2023-12-29</span></span>.</cite></span>
</li>
<li id="cite_note-161"><span class="mw-cite-backlink"><b><a href="#cite_ref-161">^</a></b></span> <span class="reference-text"><cite id="CITEREFMehta2024" class="citation web cs1">Mehta, Sourabh (2024-07-03). <a rel="nofollow" class="external text" href="https://adasci.org/how-much-energy-do-llms-consume-unveiling-the-power-behind-ai/">"How Much Energy Do LLMs Consume? Unveiling the Power Behind AI"</a>. <i>Association of Data Scientists</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-01-27</span></span>.</cite></span>
</li>
<li id="cite_note-162"><span class="mw-cite-backlink"><b><a href="#cite_ref-162">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.npr.org/2024/12/09/nx-s1-5171063/artificial-intelligence-wants-to-go-nuclear-will-it-work">"Artificial Intelligence wants to go nuclear. Will it work?"</a>. <i>NPR</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-01-27</span></span>.</cite></span>
</li>
<li id="cite_note-163"><span class="mw-cite-backlink"><b><a href="#cite_ref-163">^</a></b></span> <span class="reference-text"><cite id="CITEREFRoy,_Dareen2024" class="citation web cs1">Roy, Dareen (December 19, 2024). <a rel="nofollow" class="external text" href="https://www.reuters.com/technology/artificial-intelligence/ais-energy-hunger-fuels-geothermal-startups-natgas-rivalry-clouds-future-2024-12-19/">"AI's energy hunger fuels geothermal startups but natgas rivalry clouds future"</a>. <i>Reuters</i>.</cite></span>
</li>
<li id="cite_note-164"><span class="mw-cite-backlink"><b><a href="#cite_ref-164">^</a></b></span> <span class="reference-text"><cite id="CITEREFKosmynaHauptmannYuanSitu2025" class="citation arxiv cs1">Kosmyna, Nataliya; Hauptmann, Eugene; Yuan, Ye Tong; Situ, Jessica; Liao, Xian-Hao; Beresnitzky, Ashly Vivian; Braunstein, Iris; Maes, Pattie (June 10, 2025). "Your Brain on ChatGPT: Accumulation of Cognitive Debt when Using an AI Assistant for Essay Writing Task". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2506.08872">2506.08872</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.AI">cs.AI</a>].</cite></span>
</li>
<li id="cite_note-165"><span class="mw-cite-backlink"><b><a href="#cite_ref-165">^</a></b></span> <span class="reference-text"><cite id="CITEREFZao-Sanders2024" class="citation news cs1">Zao-Sanders, Marc (2024-03-19). <a rel="nofollow" class="external text" href="https://hbr.org/2024/03/how-people-are-really-using-genai">"How People Are Really Using GenAI"</a>. <i>Harvard Business Review</i>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/0017-8012">0017-8012</a><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-08-10</span></span>.</cite></span>
</li>
<li id="cite_note-166"><span class="mw-cite-backlink"><b><a href="#cite_ref-166">^</a></b></span> <span class="reference-text"><cite id="CITEREFRousmaniereZhangLiShah2025" class="citation journal cs1">Rousmaniere, Tony; Zhang, Yimeng; Li, Xu; Shah, Siddharth (2025-07-21). <a rel="nofollow" class="external text" href="https://doi.apa.org/doi/10.1037/pri0000292">"Large language models as mental health resources: Patterns of use in the United States"</a>. <i>Practice Innovations</i>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1037%2Fpri0000292">10.1037/pri0000292</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/2377-8903">2377-8903</a>.</cite></span>
</li>
<li id="cite_note-167"><span class="mw-cite-backlink"><b><a href="#cite_ref-167">^</a></b></span> <span class="reference-text"><cite id="CITEREFJiZhangYangAnaniadou2023" class="citation arxiv cs1">Ji, Shaoxiong; Zhang, Tianlin; Yang, Kailai; Ananiadou, Sophia; Cambria, Erik (2023-12-17). "Rethinking Large Language Models in Mental Health Applications". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2311.11267">2311.11267</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-168"><span class="mw-cite-backlink"><b><a href="#cite_ref-168">^</a></b></span> <span class="reference-text"><cite id="CITEREFMooreGrabbAgnewKlyman2025" class="citation book cs1">Moore, Jared; Grabb, Declan; Agnew, William; Klyman, Kevin; Chancellor, Stevie; Ong, Desmond C.; Haber, Nick (2025-04-25). "Expressing stigma and inappropriate responses prevents LLMS from safely replacing mental health providers". <i>Proceedings of the 2025 ACM Conference on Fairness, Accountability, and Transparency</i>. pp.&nbsp;<span class="nowrap">599–</span>627. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2504.18412">2504.18412</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1145%2F3715275.3732039">10.1145/3715275.3732039</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>979-8-4007-1482-5</bdi>.</cite></span>
</li>
<li id="cite_note-169"><span class="mw-cite-backlink"><b><a href="#cite_ref-169">^</a></b></span> <span class="reference-text"><cite id="CITEREFGrabbLamparthVasan2024" class="citation arxiv cs1">Grabb, Declan; Lamparth, Max; Vasan, Nina (2024-08-14). "Risks from Language Models for Automated Mental Healthcare: Ethics and Structure for Implementation". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2406.11852">2406.11852</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CY">cs.CY</a>].</cite></span>
</li>
<li id="cite_note-170"><span class="mw-cite-backlink"><b><a href="#cite_ref-170">^</a></b></span> <span class="reference-text"><cite id="CITEREFMcBainCantorZhangBaker2025" class="citation journal cs1">McBain, Ryan K.; Cantor, Jonathan H.; Zhang, Li Ang; Baker, Olesya; Zhang, Fang; Halbisen, Alyssa; Kofner, Aaron; Breslau, Joshua; Stein, Bradley; Mehrotra, Ateev; Yu, Hao (2025-03-05). <a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11928068">"Competency of Large Language Models in Evaluating Appropriate Responses to Suicidal Ideation: Comparative Study"</a>. <i>Journal of Medical Internet Research</i>. <b>27</b> (1): e67891. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.2196%2F67891">10.2196/67891</a></span>. <a href="PMC_(identifier)" class="mw-redirect" title="PMC (identifier)">PMC</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11928068">11928068</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/40053817">40053817</a>.</cite></span>
</li>
</ol></div></div>
<div class="mw-heading mw-heading2"><h2 id="Further_reading">Further reading</h2></div>
<ul><li><a href="Dan_Jurafsky" title="Dan Jurafsky">Jurafsky, Dan</a>, Martin, James. H. <a rel="nofollow" class="external text" href="https://web.stanford.edu/~jurafsky/slp3/ed3book_jan72023.pdf"><i>Speech and Language Processing: An Introduction to Natural Language Processing, Computational Linguistics, and Speech Recognition</i></a>, 3rd Edition draft, 2023.</li>
<li><cite id="CITEREFYinFuZhaoLi2024" class="citation journal cs1">Yin, Shukang; Fu, Chaoyou; Zhao, Sirui; Li, Ke; Sun, Xing; Xu, Tong; Chen, Enhong (2024). <a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11645129">"A Survey on Multimodal Large Language Models"</a>. <i>National Science Review</i>. <b>11</b> (12): nwae403. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2306.13549">2306.13549</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1093%2Fnsr%2Fnwae403">10.1093/nsr/nwae403</a>. <a href="PMC_(identifier)" class="mw-redirect" title="PMC (identifier)">PMC</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11645129">11645129</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/39679213">39679213</a>.</cite></li>
<li><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://aiindex.stanford.edu/report/">"AI Index Report 2024 – Artificial Intelligence Index"</a>. <i>aiindex.stanford.edu</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-05-05</span></span>.</cite></li>
<li><cite id="CITEREFFrank2023" class="citation journal cs1">Frank, Michael C. (27 June 2023). <span class="id-lock-subscription" title="Paid subscription required"><a rel="nofollow" class="external text" href="https://www.nature.com/articles/s44159-023-00211-x">"Baby steps in evaluating the capacities of large language models"</a></span>. <i>Nature Reviews Psychology</i>. <b>2</b> (8): <span class="nowrap">451–</span>452. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1038%2Fs44159-023-00211-x">10.1038/s44159-023-00211-x</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/2731-0574">2731-0574</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:259713140">259713140</a><span class="reference-accessdate">. Retrieved <span class="nowrap">2 July</span> 2023</span>.</cite></li></ul>
<div class="navbox-styles"><style data-mw-deduplicate="TemplateStyles:r1236075235">
/* start https://en.wikipedia.org/ */


.mw-parser-output .navbox{box-sizing:border-box;border:1px solid #a2a9b1;width:100%;clear:both;font-size:88%;text-align:center;padding:1px;margin:1em auto 0}.mw-parser-output .navbox .navbox{margin-top:0}.mw-parser-output .navbox+.navbox,.mw-parser-output .navbox+.navbox-styles+.navbox{margin-top:-1px}.mw-parser-output .navbox-inner,.mw-parser-output .navbox-subgroup{width:100%}.mw-parser-output .navbox-group,.mw-parser-output .navbox-title,.mw-parser-output .navbox-abovebelow{padding:0.25em 1em;line-height:1.5em;text-align:center}.mw-parser-output .navbox-group{white-space:nowrap;text-align:right}.mw-parser-output .navbox,.mw-parser-output .navbox-subgroup{background-color:#fdfdfd}.mw-parser-output .navbox-list{line-height:1.5em;border-color:#fdfdfd}.mw-parser-output .navbox-list-with-group{text-align:left;border-left-width:2px;border-left-style:solid}.mw-parser-output tr+tr>.navbox-abovebelow,.mw-parser-output tr+tr>.navbox-group,.mw-parser-output tr+tr>.navbox-image,.mw-parser-output tr+tr>.navbox-list{border-top:2px solid #fdfdfd}.mw-parser-output .navbox-title{background-color:#ccf}.mw-parser-output .navbox-abovebelow,.mw-parser-output .navbox-group,.mw-parser-output .navbox-subgroup .navbox-title{background-color:#ddf}.mw-parser-output .navbox-subgroup .navbox-group,.mw-parser-output .navbox-subgroup .navbox-abovebelow{background-color:#e6e6ff}.mw-parser-output .navbox-even{background-color:#f7f7f7}.mw-parser-output .navbox-odd{background-color:transparent}.mw-parser-output .navbox .hlist td dl,.mw-parser-output .navbox .hlist td ol,.mw-parser-output .navbox .hlist td ul,.mw-parser-output .navbox td.hlist dl,.mw-parser-output .navbox td.hlist ol,.mw-parser-output .navbox td.hlist ul{padding:0.125em 0}.mw-parser-output .navbox .navbar{display:block;font-size:100%}.mw-parser-output .navbox-title .navbar{float:left;text-align:left;margin-right:0.5em}body.skin--responsive .mw-parser-output .navbox-image img{max-width:none!important}@media print{body.ns-0 .mw-parser-output .navbox{display:none!important}}


/* end https://en.wikipedia.org/ */
</style></div><div role="navigation" class="navbox" aria-labelledby="Natural_language_processing454" style="padding:3px"><table class="nowraplinks hlist mw-collapsible autocollapse navbox-inner" style="border-spacing:0;background:transparent;color:inherit"><tbody><tr><th scope="col" class="navbox-title" colspan="2"><div id="Natural_language_processing454" style="font-size:114%;margin:0 4em"><a href="Natural_language_processing" title="Natural language processing">Natural language processing</a></div></th></tr><tr><th scope="row" class="navbox-group" style="width:1%">General terms</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="AI-complete" title="AI-complete">AI-complete</a></li>
<li><a href="Bag-of-words_model" title="Bag-of-words model">Bag-of-words</a></li>
<li><a href="N-gram" title="N-gram"><i>n</i>-gram</a>
<ul><li><a href="Bigram" title="Bigram">Bigram</a></li>
<li><a href="Trigram" title="Trigram">Trigram</a></li></ul></li>
<li><a href="Computational_linguistics" title="Computational linguistics">Computational linguistics</a></li>
<li><a href="Natural_language_understanding" title="Natural language understanding">Natural language understanding</a></li>
<li><a href="Stop_word" title="Stop word">Stop words</a></li>
<li><a href="Text_processing" title="Text processing">Text processing</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Text_mining" title="Text mining">Text analysis</a></th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Argument_mining" title="Argument mining">Argument mining</a></li>
<li><a href="Collocation_extraction" title="Collocation extraction">Collocation extraction</a></li>
<li><a href="Concept_mining" title="Concept mining">Concept mining</a></li>
<li><a href="Coreference#Coreference_resolution" title="Coreference">Coreference resolution</a></li>
<li><a href="Deep_linguistic_processing" title="Deep linguistic processing">Deep linguistic processing</a></li>
<li><a href="Distant_reading" title="Distant reading">Distant reading</a></li>
<li><a href="Information_extraction" title="Information extraction">Information extraction</a></li>
<li><a href="Named-entity_recognition" title="Named-entity recognition">Named-entity recognition</a></li>
<li><a href="Ontology_learning" title="Ontology learning">Ontology learning</a></li>
<li><a href="Parsing" title="Parsing">Parsing</a>
<ul><li><a href="Semantic_parsing" title="Semantic parsing">Semantic parsing</a></li>
<li><a href="Syntactic_parsing_(computational_linguistics)" title="Syntactic parsing (computational linguistics)">Syntactic parsing</a></li></ul></li>
<li><a href="Part-of-speech_tagging" title="Part-of-speech tagging">Part-of-speech tagging</a></li>
<li><a href="Semantic_analysis_(machine_learning)" title="Semantic analysis (machine learning)">Semantic analysis</a></li>
<li><a href="Semantic_role_labeling" title="Semantic role labeling">Semantic role labeling</a></li>
<li><a href="Semantic_decomposition_(natural_language_processing)" title="Semantic decomposition (natural language processing)">Semantic decomposition</a></li>
<li><a href="Semantic_similarity" title="Semantic similarity">Semantic similarity</a></li>
<li><a href="Sentiment_analysis" title="Sentiment analysis">Sentiment analysis</a></li></ul>
<ul><li><a href="Terminology_extraction" title="Terminology extraction">Terminology extraction</a></li>
<li><a href="Text_mining" title="Text mining">Text mining</a></li>
<li><a href="Textual_entailment" title="Textual entailment">Textual entailment</a></li>
<li><a href="Truecasing" title="Truecasing">Truecasing</a></li>
<li><a href="Word-sense_disambiguation" title="Word-sense disambiguation">Word-sense disambiguation</a></li>
<li><a href="Word-sense_induction" title="Word-sense induction">Word-sense induction</a></li></ul>
</div><table class="nowraplinks navbox-subgroup" style="border-spacing:0"><tbody><tr><th id="Text_segmentation21" scope="row" class="navbox-group" style="width:1%"><a href="Text_segmentation" title="Text segmentation">Text segmentation</a></th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Compound-term_processing" title="Compound-term processing">Compound-term processing</a></li>
<li><a href="Lemmatisation" class="mw-redirect" title="Lemmatisation">Lemmatisation</a></li>
<li><a href="Lexical_analysis" title="Lexical analysis">Lexical analysis</a></li>
<li><a href="Shallow_parsing" title="Shallow parsing">Text chunking</a></li>
<li><a href="Stemming" title="Stemming">Stemming</a></li>
<li><a href="Sentence_boundary_disambiguation" title="Sentence boundary disambiguation">Sentence segmentation</a></li>
<li><a href="Word#Word_boundaries" title="Word">Word segmentation</a></li></ul>
</div></td></tr></tbody></table><div>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Automatic_summarization" title="Automatic summarization">Automatic summarization</a></th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Multi-document_summarization" title="Multi-document summarization">Multi-document summarization</a></li>
<li><a href="Sentence_extraction" title="Sentence extraction">Sentence extraction</a></li>
<li><a href="Text_simplification" title="Text simplification">Text simplification</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Machine_translation" title="Machine translation">Machine translation</a></th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Computer-assisted_translation" title="Computer-assisted translation">Computer-assisted</a></li>
<li><a href="Example-based_machine_translation" title="Example-based machine translation">Example-based</a></li>
<li><a href="Rule-based_machine_translation" title="Rule-based machine translation">Rule-based</a></li>
<li><a href="Statistical_machine_translation" title="Statistical machine translation">Statistical</a></li>
<li><a href="Transfer-based_machine_translation" title="Transfer-based machine translation">Transfer-based</a></li>
<li><a href="Neural_machine_translation" title="Neural machine translation">Neural</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Distributional_semantics" title="Distributional semantics">Distributional semantics</a> models</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="BERT_(language_model)" title="BERT (language model)">BERT</a></li>
<li><a href="Document-term_matrix" title="Document-term matrix">Document-term matrix</a></li>
<li><a href="Explicit_semantic_analysis" title="Explicit semantic analysis">Explicit semantic analysis</a></li>
<li><a href="FastText" title="FastText">fastText</a></li>
<li><a href="GloVe" title="GloVe">GloVe</a></li>
<li><a href="Language_model" title="Language model">Language model</a> ()</li>
<li><a href="Latent_semantic_analysis" title="Latent semantic analysis">Latent semantic analysis</a></li>
<li><a href="Seq2seq" title="Seq2seq">Seq2seq</a></li>
<li><a href="Word_embedding" title="Word embedding">Word embedding</a></li>
<li><a href="Word2vec" title="Word2vec">Word2vec</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Language_resource" title="Language resource">Language resources</a>,<br>datasets and corpora</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em"></div><table class="nowraplinks navbox-subgroup" style="border-spacing:0"><tbody><tr><th scope="row" class="navbox-group" style="width:1%">Types and<br>standards</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Corpus_linguistics" title="Corpus linguistics">Corpus linguistics</a></li>
<li><a href="Lexical_resource" title="Lexical resource">Lexical resource</a></li>
<li><a href="Linguistic_Linked_Open_Data" title="Linguistic Linked Open Data">Linguistic Linked Open Data</a></li>
<li><a href="Machine-readable_dictionary" title="Machine-readable dictionary">Machine-readable dictionary</a></li>
<li><a href="Parallel_text" title="Parallel text">Parallel text</a></li>
<li><a href="PropBank" title="PropBank">PropBank</a></li>
<li><a href="Semantic_network" title="Semantic network">Semantic network</a></li>
<li><a href="Simple_Knowledge_Organization_System" title="Simple Knowledge Organization System">Simple Knowledge Organization System</a></li>
<li><a href="Speech_corpus" title="Speech corpus">Speech corpus</a></li>
<li><a href="Text_corpus" title="Text corpus">Text corpus</a></li>
<li><a href="Thesaurus_(information_retrieval)" title="Thesaurus (information retrieval)">Thesaurus (information retrieval)</a></li>
<li><a href="Treebank" title="Treebank">Treebank</a></li>
<li><a href="Universal_Dependencies" title="Universal Dependencies">Universal Dependencies</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Data</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="BabelNet" title="BabelNet">BabelNet</a></li>
<li><a href="Bank_of_English" title="Bank of English">Bank of English</a></li>
<li><a href="DBpedia" title="DBpedia">DBpedia</a></li>
<li><a href="FrameNet" title="FrameNet">FrameNet</a></li>
<li><a href="Google_Ngram_Viewer" class="mw-redirect" title="Google Ngram Viewer">Google Ngram Viewer</a></li>
<li><a href="UBY" title="UBY">UBY</a></li>
<li><a href="WordNet" title="WordNet">WordNet</a></li>
<li><a href="Wikidata" title="Wikidata">Wikidata</a></li></ul>
</div></td></tr></tbody></table><div></div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Automatic_identification_and_data_capture" title="Automatic identification and data capture">Automatic identification<br>and data capture</a></th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Speech_recognition" title="Speech recognition">Speech recognition</a></li>
<li><a href="Speech_segmentation" title="Speech segmentation">Speech segmentation</a></li>
<li><a href="Speech_synthesis" title="Speech synthesis">Speech synthesis</a></li>
<li><a href="Natural_language_generation" title="Natural language generation">Natural language generation</a></li>
<li><a href="Optical_character_recognition" title="Optical character recognition">Optical character recognition</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Topic_model" title="Topic model">Topic model</a></th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Document_classification" title="Document classification">Document classification</a></li>
<li><a href="Latent_Dirichlet_allocation" title="Latent Dirichlet allocation">Latent Dirichlet allocation</a></li>
<li><a href="Pachinko_allocation" title="Pachinko allocation">Pachinko allocation</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Computer-assisted_reviewing" title="Computer-assisted reviewing">Computer-assisted<br>reviewing</a></th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Automated_essay_scoring" title="Automated essay scoring">Automated essay scoring</a></li>
<li><a href="Concordancer" title="Concordancer">Concordancer</a></li>
<li><a href="Grammar_checker" title="Grammar checker">Grammar checker</a></li>
<li><a href="Predictive_text" title="Predictive text">Predictive text</a></li>
<li><a href="Pronunciation_assessment" title="Pronunciation assessment">Pronunciation assessment</a></li>
<li><a href="Spell_checker" title="Spell checker">Spell checker</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Natural-language_user_interface" title="Natural-language user interface">Natural language<br>user interface</a></th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Chatbot" title="Chatbot">Chatbot</a></li>
<li><a href="Interactive_fiction" title="Interactive fiction">Interactive fiction</a></li>
<li><a href="Question_answering" title="Question answering">Question answering</a></li>
<li><a href="Virtual_assistant" title="Virtual assistant">Virtual assistant</a></li>
<li><a href="Voice_user_interface" title="Voice user interface">Voice user interface</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Related</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Formal_semantics_(natural_language)" title="Formal semantics (natural language)">Formal semantics</a></li>
<li><a href="Hallucination_(artificial_intelligence)" title="Hallucination (artificial intelligence)">Hallucination</a></li>
<li><a href="Natural_Language_Toolkit" title="Natural Language Toolkit">Natural Language Toolkit</a></li>
<li><a href="SpaCy" title="SpaCy">spaCy</a></li></ul>
</div></td></tr></tbody></table></div>
<div class="navbox-styles"></div><div role="navigation" class="navbox" aria-labelledby="Artificial_intelligence_(AI)426" style="padding:3px"><table class="nowraplinks hlist mw-collapsible autocollapse navbox-inner" style="border-spacing:0;background:transparent;color:inherit"><tbody><tr><th scope="col" class="navbox-title" colspan="2"><div id="Artificial_intelligence_(AI)426" style="font-size:114%;margin:0 4em"><a href="Artificial_intelligence" title="Artificial intelligence">Artificial intelligence</a> (AI)</div></th></tr><tr><td class="navbox-abovebelow" colspan="2"><div>
<ul><li><a href="History_of_artificial_intelligence" title="History of artificial intelligence">History</a>
<ul><li><a href="Timeline_of_artificial_intelligence" title="Timeline of artificial intelligence">timeline</a></li></ul></li>
<li><a href="List_of_artificial_intelligence_companies" title="List of artificial intelligence companies">Companies</a></li>
<li><a href="List_of_artificial_intelligence_projects" title="List of artificial intelligence projects">Projects</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Concepts</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Parameter" title="Parameter">Parameter</a>
<ul><li><a href="Hyperparameter_(machine_learning)" title="Hyperparameter (machine learning)">Hyperparameter</a></li></ul></li>
<li><a href="Loss_functions_for_classification" title="Loss functions for classification">Loss functions</a></li>
<li><a href="Regression_analysis" title="Regression analysis">Regression</a>
<ul><li><a href="Bias%E2%80%93variance_tradeoff" title="Bias–variance tradeoff">Bias–variance tradeoff</a></li>
<li><a href="Double_descent" title="Double descent">Double descent</a></li>
<li><a href="Overfitting" title="Overfitting">Overfitting</a></li></ul></li>
<li><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></li>
<li><a href="Gradient_descent" title="Gradient descent">Gradient descent</a>
<ul><li><a href="Stochastic_gradient_descent" title="Stochastic gradient descent">SGD</a></li>
<li><a href="Quasi-Newton_method" title="Quasi-Newton method">Quasi-Newton method</a></li>
<li><a href="Conjugate_gradient_method" title="Conjugate gradient method">Conjugate gradient method</a></li></ul></li>
<li><a href="Backpropagation" title="Backpropagation">Backpropagation</a></li>
<li><a href="Attention_(machine_learning)" title="Attention (machine learning)">Attention</a></li>
<li><a href="Convolution" title="Convolution">Convolution</a></li>
<li><a href="Normalization_(machine_learning)" title="Normalization (machine learning)">Normalization</a>
<ul><li><a href="Batch_normalization" title="Batch normalization">Batchnorm</a></li></ul></li>
<li><a href="Activation_function" title="Activation function">Activation</a>
<ul><li><a href="Softmax_function" title="Softmax function">Softmax</a></li>
<li><a href="Sigmoid_function" title="Sigmoid function">Sigmoid</a></li>
<li><a href="Rectifier_(neural_networks)" title="Rectifier (neural networks)">Rectifier</a></li></ul></li>
<li><a href="Gating_mechanism" title="Gating mechanism">Gating</a></li>
<li><a href="Weight_initialization" title="Weight initialization">Weight initialization</a></li>
<li><a href="Regularization_(mathematics)" title="Regularization (mathematics)">Regularization</a></li>
<li><a href="Training%2C_validation%2C_and_test_data_sets" title="Training, validation, and test data sets">Datasets</a>
<ul><li><a href="Data_augmentation" title="Data augmentation">Augmentation</a></li></ul></li>
<li><a href="Prompt_engineering" title="Prompt engineering">Prompt engineering</a></li>
<li><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a>
<ul><li><a href="Q-learning" title="Q-learning">Q-learning</a></li>
<li><a href="State%E2%80%93action%E2%80%93reward%E2%80%93state%E2%80%93action" title="State–action–reward–state–action">SARSA</a></li>
<li><a href="Imitation_learning" title="Imitation learning">Imitation</a></li>
<li><a href="Policy_gradient_method" title="Policy gradient method">Policy gradient</a></li></ul></li>
<li><a href="Diffusion_process" title="Diffusion process">Diffusion</a></li>
<li><a href="Latent_diffusion_model" title="Latent diffusion model">Latent diffusion model</a></li>
<li><a href="Autoregressive_model" title="Autoregressive model">Autoregression</a></li>
<li><a href="Adversarial_machine_learning" title="Adversarial machine learning">Adversary</a></li>
<li><a href="Retrieval-augmented_generation" title="Retrieval-augmented generation">RAG</a></li>
<li><a href="Uncanny_valley" title="Uncanny valley">Uncanny valley</a></li>
<li><a href="Reinforcement_learning_from_human_feedback" title="Reinforcement learning from human feedback">RLHF</a></li>
<li><a href="Self-supervised_learning" title="Self-supervised learning">Self-supervised learning</a></li>
<li><a href="Reflection_(artificial_intelligence)" class="mw-redirect" title="Reflection (artificial intelligence)">Reflection</a></li>
<li><a href="Recursive_self-improvement" title="Recursive self-improvement">Recursive self-improvement</a></li>
<li><a href="Hallucination_(artificial_intelligence)" title="Hallucination (artificial intelligence)">Hallucination</a></li>
<li><a href="Word_embedding" title="Word embedding">Word embedding</a></li>
<li><a href="Vibe_coding" title="Vibe coding">Vibe coding</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Applications</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Machine_learning" title="Machine learning">Machine learning</a>
<ul><li><a href="Prompt_engineering#In-context_learning" title="Prompt engineering">In-context learning</a></li></ul></li>
<li><a href="Neural_network_(machine_learning)" title="Neural network (machine learning)">Artificial neural network</a>
<ul><li><a href="Deep_learning" title="Deep learning">Deep learning</a></li></ul></li>
<li><a href="Language_model" title="Language model">Language model</a>
<ul>
<li><a href="Neural_machine_translation" title="Neural machine translation">NMT</a></li></ul></li>
<li><a href="Reasoning_language_model" title="Reasoning language model">Reasoning language model</a></li>
<li><a href="Model_Context_Protocol" title="Model Context Protocol">Model Context Protocol</a></li>
<li><a href="Intelligent_agent" title="Intelligent agent">Intelligent agent</a></li>
<li><a href="Artificial_human_companion" title="Artificial human companion">Artificial human companion</a></li>
<li><a href="Humanity's_Last_Exam" title="Humanity's Last Exam">Humanity's Last Exam</a></li>
<li><a href="Artificial_general_intelligence" title="Artificial general intelligence">Artificial general intelligence (AGI)</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Implementations</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em"></div><table class="nowraplinks navbox-subgroup" style="border-spacing:0"><tbody><tr><th scope="row" class="navbox-group" style="width:1%">Audio–visual</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="AlexNet" title="AlexNet">AlexNet</a></li>
<li><a href="WaveNet" title="WaveNet">WaveNet</a></li>
<li><a href="Human_image_synthesis" title="Human image synthesis">Human image synthesis</a></li>
<li><a href="Handwriting_recognition" title="Handwriting recognition">HWR</a></li>
<li><a href="Optical_character_recognition" title="Optical character recognition">OCR</a></li>
<li><a href="Computer_vision" title="Computer vision">Computer vision</a></li>
<li><a href="Deep_learning_speech_synthesis" title="Deep learning speech synthesis">Speech synthesis</a>
<ul><li><a href="15.ai" title="15.ai">15.ai</a></li>
<li><a href="ElevenLabs" title="ElevenLabs">ElevenLabs</a></li></ul></li>
<li><a href="Speech_recognition" title="Speech recognition">Speech recognition</a>
<ul><li><a href="Whisper_(speech_recognition_system)" title="Whisper (speech recognition system)">Whisper</a></li></ul></li>
<li><a href="Facial_recognition_system" title="Facial recognition system">Facial recognition</a></li>
<li><a href="AlphaFold" title="AlphaFold">AlphaFold</a></li>
<li><a href="Text-to-image_model" title="Text-to-image model">Text-to-image models</a>
<ul><li><a href="Aurora_(text-to-image_model)" class="mw-redirect" title="Aurora (text-to-image model)">Aurora</a></li>
<li><a href="DALL-E" title="DALL-E">DALL-E</a></li>
<li><a href="Adobe_Firefly" title="Adobe Firefly">Firefly</a></li>
<li><a href="Flux_(text-to-image_model)" title="Flux (text-to-image model)">Flux</a></li>
<li><a href="Ideogram_(text-to-image_model)" title="Ideogram (text-to-image model)">Ideogram</a></li>
<li><a href="Imagen_(text-to-image_model)" title="Imagen (text-to-image model)">Imagen</a></li>
<li><a href="Midjourney" title="Midjourney">Midjourney</a></li>
<li><a href="Recraft" title="Recraft">Recraft</a></li>
<li><a href="Stable_Diffusion" title="Stable Diffusion">Stable Diffusion</a></li></ul></li>
<li><a href="Text-to-video_model" title="Text-to-video model">Text-to-video models</a>
<ul><li><a href="Dream_Machine_(text-to-video_model)" title="Dream Machine (text-to-video model)">Dream Machine</a></li>
<li><a href="Runway_(company)#Services_and_technologies" title="Runway (company)">Runway Gen</a></li>
<li><a href="MiniMax_(company)#Hailuo_AI" title="MiniMax (company)">Hailuo AI</a></li>
<li><a href="Kling_(text-to-video_model)" class="mw-redirect" title="Kling (text-to-video model)">Kling</a></li>
<li><a href="Sora_(text-to-video_model)" title="Sora (text-to-video model)">Sora</a></li>
<li><a href="Veo_(text-to-video_model)" title="Veo (text-to-video model)">Veo</a></li></ul></li>
<li><a href="Music_and_artificial_intelligence" title="Music and artificial intelligence">Music generation</a>
<ul><li><a href="Riffusion" title="Riffusion">Riffusion</a></li>
<li><a href="Suno_AI" title="Suno AI">Suno AI</a></li>
<li><a href="Udio" title="Udio">Udio</a></li></ul></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Text</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Word2vec" title="Word2vec">Word2vec</a></li>
<li><a href="Seq2seq" title="Seq2seq">Seq2seq</a></li>
<li><a href="GloVe" title="GloVe">GloVe</a></li>
<li><a href="BERT_(language_model)" title="BERT (language model)">BERT</a></li>
<li><a href="T5_(language_model)" title="T5 (language model)">T5</a></li>
<li><a href="Llama_(language_model)" title="Llama (language model)">Llama</a></li>
<li><a href="Chinchilla_(language_model)" title="Chinchilla (language model)">Chinchilla AI</a></li>
<li><a href="PaLM" title="PaLM">PaLM</a></li>
<li><a href="Generative_pre-trained_transformer" title="Generative pre-trained transformer">GPT</a>
<ul><li><a href="GPT-1" title="GPT-1">1</a></li>
<li><a href="GPT-2" title="GPT-2">2</a></li>
<li><a href="GPT-3" title="GPT-3">3</a></li>
<li><a href="GPT-J" title="GPT-J">J</a></li>
<li><a href="ChatGPT" title="ChatGPT">ChatGPT</a></li>
<li><a href="GPT-4" title="GPT-4">4</a></li>
<li><a href="GPT-4o" title="GPT-4o">4o</a></li>
<li><a href="OpenAI_o1" title="OpenAI o1">o1</a></li>
<li><a href="OpenAI_o3" title="OpenAI o3">o3</a></li>
<li><a href="GPT-4.5" title="GPT-4.5">4.5</a></li>
<li><a href="GPT-4.1" title="GPT-4.1">4.1</a></li>
<li><a href="OpenAI_o4-mini" title="OpenAI o4-mini">o4-mini</a></li>
<li><a href="GPT-5" title="GPT-5">5</a></li></ul></li>
<li><a href="Claude_(language_model)" title="Claude (language model)">Claude</a></li>
<li><a href="Gemini_(language_model)" title="Gemini (language model)">Gemini</a>
<ul><li><a href="Gemini_(chatbot)" title="Gemini (chatbot)">chatbot</a></li></ul></li>
<li><a href="Grok_(chatbot)" title="Grok (chatbot)">Grok</a></li>
<li><a href="LaMDA" title="LaMDA">LaMDA</a></li>
<li><a href="BLOOM_(language_model)" title="BLOOM (language model)">BLOOM</a></li>
<li><a href="DBRX" title="DBRX">DBRX</a></li>
<li><a href="Project_Debater" title="Project Debater">Project Debater</a></li>
<li><a href="IBM_Watson" title="IBM Watson">IBM Watson</a></li>
<li><a href="IBM_Watsonx" title="IBM Watsonx">IBM Watsonx</a></li>
<li><a href="IBM_Granite" title="IBM Granite">Granite</a></li>
<li><a href="Huawei_PanGu" title="Huawei PanGu">PanGu-Σ</a></li>
<li><a href="DeepSeek_(chatbot)" title="DeepSeek (chatbot)">DeepSeek</a></li>
<li><a href="Qwen" title="Qwen">Qwen</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Decisional</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="AlphaGo" title="AlphaGo">AlphaGo</a></li>
<li><a href="AlphaZero" title="AlphaZero">AlphaZero</a></li>
<li><a href="OpenAI_Five" title="OpenAI Five">OpenAI Five</a></li>
<li><a href="Self-driving_car" title="Self-driving car">Self-driving car</a></li>
<li><a href="MuZero" title="MuZero">MuZero</a></li>
<li><a href="Action_selection" title="Action selection">Action selection</a>
<ul><li><a href="AutoGPT" title="AutoGPT">AutoGPT</a></li></ul></li>
<li><a href="Robot_control" title="Robot control">Robot control</a></li></ul>
</div></td></tr></tbody></table><div></div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">People</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Alan_Turing" title="Alan Turing">Alan Turing</a></li>
<li><a href="Warren_Sturgis_McCulloch" title="Warren Sturgis McCulloch">Warren Sturgis McCulloch</a></li>
<li><a href="Walter_Pitts" title="Walter Pitts">Walter Pitts</a></li>
<li><a href="John_von_Neumann" title="John von Neumann">John von Neumann</a></li>
<li><a href="Claude_Shannon" title="Claude Shannon">Claude Shannon</a></li>
<li><a href="Shun'ichi_Amari" title="Shun'ichi Amari">Shun'ichi Amari</a></li>
<li><a href="Kunihiko_Fukushima" title="Kunihiko Fukushima">Kunihiko Fukushima</a></li>
<li><a href="Takeo_Kanade" title="Takeo Kanade">Takeo Kanade</a></li>
<li><a href="Marvin_Minsky" title="Marvin Minsky">Marvin Minsky</a></li>
<li><a href="John_McCarthy_(computer_scientist)" title="John McCarthy (computer scientist)">John McCarthy</a></li>
<li><a href="Nathaniel_Rochester_(computer_scientist)" title="Nathaniel Rochester (computer scientist)">Nathaniel Rochester</a></li>
<li><a href="Allen_Newell" title="Allen Newell">Allen Newell</a></li>
<li><a href="Cliff_Shaw" title="Cliff Shaw">Cliff Shaw</a></li>
<li><a href="Herbert_A._Simon" title="Herbert A. Simon">Herbert A. Simon</a></li>
<li><a href="Oliver_Selfridge" title="Oliver Selfridge">Oliver Selfridge</a></li>
<li><a href="Frank_Rosenblatt" title="Frank Rosenblatt">Frank Rosenblatt</a></li>
<li><a href="Bernard_Widrow" title="Bernard Widrow">Bernard Widrow</a></li>
<li><a href="Joseph_Weizenbaum" title="Joseph Weizenbaum">Joseph Weizenbaum</a></li>
<li><a href="Seymour_Papert" title="Seymour Papert">Seymour Papert</a></li>
<li><a href="Seppo_Linnainmaa" title="Seppo Linnainmaa">Seppo Linnainmaa</a></li>
<li><a href="Paul_Werbos" title="Paul Werbos">Paul Werbos</a></li>
<li><a href="Geoffrey_Hinton" title="Geoffrey Hinton">Geoffrey Hinton</a></li>
<li><a href="John_Hopfield" title="John Hopfield">John Hopfield</a></li>
<li><a href="J%C3%BCrgen_Schmidhuber" title="Jürgen Schmidhuber">Jürgen Schmidhuber</a></li>
<li><a href="Yann_LeCun" title="Yann LeCun">Yann LeCun</a></li>
<li><a href="Yoshua_Bengio" title="Yoshua Bengio">Yoshua Bengio</a></li>
<li><a href="Lotfi_A._Zadeh" title="Lotfi A. Zadeh">Lotfi A. Zadeh</a></li>
<li><a href="Stephen_Grossberg" title="Stephen Grossberg">Stephen Grossberg</a></li>
<li><a href="Alex_Graves_(computer_scientist)" title="Alex Graves (computer scientist)">Alex Graves</a></li>
<li><a href="James_Goodnight" title="James Goodnight">James Goodnight</a></li>
<li><a href="Andrew_Ng" title="Andrew Ng">Andrew Ng</a></li>
<li><a href="Fei-Fei_Li" title="Fei-Fei Li">Fei-Fei Li</a></li>
<li><a href="Ilya_Sutskever" title="Ilya Sutskever">Ilya Sutskever</a></li>
<li><a href="Alex_Krizhevsky" title="Alex Krizhevsky">Alex Krizhevsky</a></li>
<li><a href="Ian_Goodfellow" title="Ian Goodfellow">Ian Goodfellow</a></li>
<li><a href="Demis_Hassabis" title="Demis Hassabis">Demis Hassabis</a></li>
<li><a href="David_Silver_(computer_scientist)" title="David Silver (computer scientist)">David Silver</a></li>
<li><a href="Andrej_Karpathy" title="Andrej Karpathy">Andrej Karpathy</a></li>
<li><a href="Ashish_Vaswani" title="Ashish Vaswani">Ashish Vaswani</a></li>
<li><a href="Noam_Shazeer" title="Noam Shazeer">Noam Shazeer</a></li>
<li><a href="Aidan_Gomez" title="Aidan Gomez">Aidan Gomez</a></li>
<li><a href="Mustafa_Suleyman" title="Mustafa Suleyman">Mustafa Suleyman</a></li>
<li><a href="Fran%C3%A7ois_Chollet" title="François Chollet">François Chollet</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Architectures</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Neural_Turing_machine" title="Neural Turing machine">Neural Turing machine</a></li>
<li><a href="Differentiable_neural_computer" title="Differentiable neural computer">Differentiable neural computer</a></li>
<li><a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">Transformer</a>
<ul><li><a href="Vision_transformer" title="Vision transformer">Vision transformer (ViT)</a></li></ul></li>
<li><a href="Recurrent_neural_network" title="Recurrent neural network">Recurrent neural network (RNN)</a></li>
<li><a href="Long_short-term_memory" title="Long short-term memory">Long short-term memory (LSTM)</a></li>
<li><a href="Gated_recurrent_unit" title="Gated recurrent unit">Gated recurrent unit (GRU)</a></li>
<li><a href="Echo_state_network" title="Echo state network">Echo state network</a></li>
<li><a href="Multilayer_perceptron" title="Multilayer perceptron">Multilayer perceptron (MLP)</a></li>
<li><a href="Convolutional_neural_network" title="Convolutional neural network">Convolutional neural network (CNN)</a></li>
<li><a href="Residual_neural_network" title="Residual neural network">Residual neural network (RNN)</a></li>
<li><a href="Highway_network" title="Highway network">Highway network</a></li>
<li><a href="Mamba_(deep_learning_architecture)" title="Mamba (deep learning architecture)">Mamba</a></li>
<li><a href="Autoencoder" title="Autoencoder">Autoencoder</a></li>
<li><a href="Variational_autoencoder" title="Variational autoencoder">Variational autoencoder (VAE)</a></li>
<li><a href="Generative_adversarial_network" title="Generative adversarial network">Generative adversarial network (GAN)</a></li>
<li><a href="Graph_neural_network" title="Graph neural network">Graph neural network (GNN)</a></li></ul>
</div></td></tr><tr><td class="navbox-abovebelow" colspan="2"><div>
<ul><li><span class="noviewer" typeof="mw:File"><span title="Category"></span></span> Category</li></ul>
</div></td></tr></tbody></table></div></div><!--htdig_noindex--><div><div class="zim-footer">
This article is issued from <a class="external text" title="Last edited on 2025-08-08" href="https://en.wikipedia.org/wiki/?title=Large_language_model&amp;oldid=1304816779">Wikipedia</a>. The text is available under <a class="external text" href="https://creativecommons.org/licenses/by-sa/4.0/deed.en">Creative Commons Attribution-Share Alike 4.0</a> unless otherwise noted. Additional terms may apply for the media files.
</div>
</div><!--/htdig_noindex--></div>
</div>
</main>
</div>
</div>
</div>

</body></html>